diff --git a/.dockerignore b/.dockerignore index 1d81d13e..a3b11f57 100644 --- a/.dockerignore +++ b/.dockerignore @@ -15,5 +15,13 @@ tests benchmarks skills proposals +# webapp/ is copied by demo/Dockerfile — ship only its source, never the +# host's node_modules (breaks cross-arch: npm ci reinstalls in-image) or the +# generated dist/. The root vouch image ignores webapp entirely. +webapp/node_modules +webapp/dist +webapp/test-results +webapp/playwright-report +webapp/.superpowers **/__pycache__ **/*.py[cod] diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 482b5eb2..bd3c7d5c 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -23,11 +23,20 @@ jobs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 + # Build the React console so the wheel can bundle it as vouch/web/console + # (hatch_build.py force-includes webapp/dist when present). + - uses: actions/setup-node@v4 + with: + node-version: "20" + - run: npm --prefix webapp ci && npm --prefix webapp run build - uses: actions/setup-python@v5 with: python-version: "3.12" - run: python -m pip install --upgrade build - - run: python -m build + # sdist is source-only; the wheel is built from the working tree (not the + # sdist) so the freshly-built, gitignored webapp/dist rides inside it. + - run: python -m build --sdist + - run: python -m build --wheel - uses: actions/upload-artifact@v7 with: name: dist diff --git a/.gitignore b/.gitignore index ee98bd96..b59e1ca0 100644 --- a/.gitignore +++ b/.gitignore @@ -23,3 +23,5 @@ build/ # and adapters/claude-code/.claude stay tracked) /web/ /.claude/ +.vouch +docs \ No newline at end of file diff --git a/.vouch/audit.log.jsonl b/.vouch/audit.log.jsonl index 3306eae3..73a27d42 100644 --- a/.vouch/audit.log.jsonl +++ b/.vouch/audit.log.jsonl @@ -7,3 +7,21 @@ {"actor":"a","created_at":"2026-05-21T06:05:54.067884Z","data":{},"dry_run":false,"event":"source.add","id":"0b451622d85246089a1d04c45d56c007","object_ids":["67478e72acfb8fac3a059143e95c95f5cc6f7e8d4dccc05fbcea8dbccb8a4eba"],"reversible":true} {"actor":"a","created_at":"2026-05-21T06:06:58.649209Z","data":{},"dry_run":false,"event":"source.add","id":"b72feb84d842486490bb7616c616a08c","object_ids":["67478e72acfb8fac3a059143e95c95f5cc6f7e8d4dccc05fbcea8dbccb8a4eba"],"reversible":true} {"actor":"a","created_at":"2026-05-21T06:07:08.973490Z","data":{"slug_hint":"vouch-enforces-a-review-gate-on-agent-writes"},"dry_run":false,"event":"proposal.claim.create","id":"9ef38f6a9d8c4e02ad3fd196c49bcc6f","object_ids":["20260521-060708-2b81cdab"],"reversible":true} +{"actor":"vouch-capture","created_at":"2026-07-10T06:51:08.777382Z","data":{"slug_hint":"session-71-file-s"},"dry_run":false,"event":"proposal.page.create","hash":"76845f95a24cc02ab36c491b8b121340d21291954e35d52a960bbebc85e30cdc","id":"d7b136387184484e8811038ab33cf68b","object_ids":["20260710-065108-54f0c393"],"prev_hash":"0000000000000000000000000000000000000000000000000000000000000000","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.234458Z","data":{"slug_hint":"scaffolded-inbounds-detection-engine-package-with-typescript"},"dry_run":false,"event":"proposal.page.create","hash":"6d65bcafe2c0f48f78ea5a8b52d76a924e49d80d9195b928cda6ae3a2dd6bb3b","id":"b1d08c345681419297ce580018ae4a6c","object_ids":["20260710-065335-f6053bee"],"prev_hash":"76845f95a24cc02ab36c491b8b121340d21291954e35d52a960bbebc85e30cdc","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.239312Z","data":{"slug_hint":"defined-core-domain-types-and-initial-type-level-test-covera"},"dry_run":false,"event":"proposal.page.create","hash":"c99d094fed530bed419a65cec72c63a515910cf0761aa69ef3bb2a76be4b5e14","id":"709f36729bd84a94a9d260d71eeeab24","object_ids":["20260710-065335-bc8ede6c"],"prev_hash":"6d65bcafe2c0f48f78ea5a8b52d76a924e49d80d9195b928cda6ae3a2dd6bb3b","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.242918Z","data":{"slug_hint":"built-the-github-adapter-layer-fetch-normalize-handle-action"},"dry_run":false,"event":"proposal.page.create","hash":"ac62f3954014810ff07fa2fc80a8852ae8c10571015b80c5318092dd2bb1ff08","id":"f4110c94f0404fc5b7be15e8cb8f6c56","object_ids":["20260710-065335-a882803f"],"prev_hash":"c99d094fed530bed419a65cec72c63a515910cf0761aa69ef3bb2a76be4b5e14","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.247145Z","data":{"slug_hint":"implemented-deterministic-scoring-module-with-unit-tests"},"dry_run":false,"event":"proposal.page.create","hash":"ceeb8483f50f9a6b383c0be2347fedce07941c7a70bc79e2eb924a7accc2d047","id":"0a718c0581484802a4c76038a1fb700f","object_ids":["20260710-065335-eb4a7247"],"prev_hash":"ac62f3954014810ff07fa2fc80a8852ae8c10571015b80c5318092dd2bb1ff08","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.250920Z","data":{"slug_hint":"wired-evaluate-and-cli-entry-points"},"dry_run":false,"event":"proposal.page.create","hash":"3fc88df87d56cbdb5ace39b6df14afba28748b0c1512e39eebfcf06509c07bd1","id":"f9a5694f054c4ce09c69f1feb50e497b","object_ids":["20260710-065335-142a3abd"],"prev_hash":"ceeb8483f50f9a6b383c0be2347fedce07941c7a70bc79e2eb924a7accc2d047","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.254094Z","data":{"slug_hint":"created-json-fixture-files-for-scenario-driven-testing"},"dry_run":false,"event":"proposal.page.create","hash":"0660c585390b276a3255617710e8105fe572ee23b4aed8aa54b519d3c93dc1a8","id":"c3f3d0583f8e4272b949705f977e41cc","object_ids":["20260710-065335-a8161813"],"prev_hash":"3fc88df87d56cbdb5ace39b6df14afba28748b0c1512e39eebfcf06509c07bd1","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.263211Z","data":{"reason":"superseded by llm narrative summary"},"dry_run":false,"event":"proposal.page.reject","hash":"270b31129cdf9096dbdc4717f7bfd1ebb6a0a4d6ade2e9df467b8989465f88dd","id":"900e8f66fcdf4babbdea394716afa54a","object_ids":["20260710-065108-54f0c393"],"prev_hash":"0660c585390b276a3255617710e8105fe572ee23b4aed8aa54b519d3c93dc1a8","reversible":true} +{"actor":"session-split","created_at":"2026-07-10T06:53:35.265122Z","data":{"dropped":0,"observations":0,"proposed":6,"truncated":false},"dry_run":false,"event":"session.split","hash":"234acadc3c644da01487935d5f117a25892e7b46fa1ca8dbc188ea3b77cc1c3d","id":"a98c8540c53b4021b1adc938def2335e","object_ids":["20260710-065335-f6053bee","20260710-065335-bc8ede6c","20260710-065335-a882803f","20260710-065335-eb4a7247","20260710-065335-142a3abd","20260710-065335-a8161813"],"prev_hash":"270b31129cdf9096dbdc4717f7bfd1ebb6a0a4d6ade2e9df467b8989465f88dd","reversible":true} +{"actor":"unknown-agent","created_at":"2026-07-10T07:24:02.966238Z","data":{"reason":null},"dry_run":false,"event":"proposal.page.approve","hash":"0a979813ec3ce3f3909e5f8519527157052ccdb85a5dea3fd12184aae1697130","id":"71fece2560d040a493b8f89534730775","object_ids":["20260710-065335-bc8ede6c","defined-core-domain-types-and-initial-type-level-test-covera"],"prev_hash":"234acadc3c644da01487935d5f117a25892e7b46fa1ca8dbc188ea3b77cc1c3d","reversible":true} +{"actor":"wiki-compiler","created_at":"2026-07-10T07:24:33.615350Z","data":{"slug_hint":"review-gated-knowledge-proposal-workflow"},"dry_run":false,"event":"proposal.page.create","hash":"dd18e31fc67152a2135b2ebb5e241b9e6a04d1fb19c4d77b2b91fbbcb5de775b","id":"fa3ba361a4fa4094ba52c07624c924b7","object_ids":["20260710-072433-ffff6671"],"prev_hash":"0a979813ec3ce3f3909e5f8519527157052ccdb85a5dea3fd12184aae1697130","reversible":true} +{"actor":"wiki-compiler","created_at":"2026-07-10T07:24:33.617853Z","data":{"slug_hint":"repository-as-knowledge-store"},"dry_run":false,"event":"proposal.page.create","hash":"8b35e84a4e0ef89e89c244568de6158f33117a1a64bc131542b3a94072291f60","id":"daa78f7818a1442496c0b69071378fbe","object_ids":["20260710-072433-123db33a"],"prev_hash":"dd18e31fc67152a2135b2ebb5e241b9e6a04d1fb19c4d77b2b91fbbcb5de775b","reversible":true} +{"actor":"unknown-agent","created_at":"2026-07-10T07:24:33.618337Z","data":{"dropped":0,"proposed":2,"proposer":"wiki-compiler"},"dry_run":false,"event":"compile.run","hash":"6aa930c42142624a0ca4753564efaaf5bfbf6e19ced80e7f0b0fd4c09ee66aed","id":"83373b1dda3c45f19748753b2bc9de88","object_ids":["20260710-072433-ffff6671","20260710-072433-123db33a"],"prev_hash":"8b35e84a4e0ef89e89c244568de6158f33117a1a64bc131542b3a94072291f60","reversible":true} +{"actor":"vouch-extractor","created_at":"2026-07-10T07:25:41.876231Z","data":{"slug_hint":"review-gated-knowledge-proposal-workflow-mentions-reviewed-k"},"dry_run":false,"event":"proposal.relation.create","hash":"f35167409cbe540e24293b348e0127e60553e171a35d32fb8b5b71c33317bfce","id":"e73a022074444ddb9e30407963f20907","object_ids":["20260710-072541-96ce439f"],"prev_hash":"6aa930c42142624a0ca4753564efaaf5bfbf6e19ced80e7f0b0fd4c09ee66aed","reversible":true} +{"actor":"unknown-agent","created_at":"2026-07-10T07:25:41.877790Z","data":{"reason":null},"dry_run":false,"event":"proposal.page.approve","hash":"1a2d24c3a512bc4c0171efb61c9631f4882477170e152aad4199e732f829fc5e","id":"34b056e066f7444d819f0486bc5a0584","object_ids":["20260710-072433-ffff6671","review-gated-knowledge-proposal-workflow"],"prev_hash":"f35167409cbe540e24293b348e0127e60553e171a35d32fb8b5b71c33317bfce","reversible":true} +{"actor":"unknown-agent","created_at":"2026-07-10T07:25:59.154361Z","data":{"reason":null},"dry_run":false,"event":"proposal.page.approve","hash":"b4cc6e9d79da47b096917cdeefd5d5b0910bda7b7bc06a7e146d8dd9d0038310","id":"b8c6719b8d44472883e40b016f8faad0","object_ids":["20260710-072433-123db33a","repository-as-knowledge-store"],"prev_hash":"1a2d24c3a512bc4c0171efb61c9631f4882477170e152aad4199e732f829fc5e","reversible":true} +{"actor":"unknown-agent","created_at":"2026-07-10T09:48:36.518125Z","data":{"reason":null},"dry_run":false,"event":"proposal.page.approve","hash":"9f2e75b6d37082ee754475878d52da597cc2cffaf42627b832adcf16a642df4c","id":"d509714619ff4e36b46e7c6360614f26","object_ids":["20260710-065335-eb4a7247","implemented-deterministic-scoring-module-with-unit-tests"],"prev_hash":"b4cc6e9d79da47b096917cdeefd5d5b0910bda7b7bc06a7e146d8dd9d0038310","reversible":true} +{"actor":"unknown-agent","created_at":"2026-07-10T09:48:48.002564Z","data":{"reason":null},"dry_run":false,"event":"proposal.page.approve","hash":"dbb2d71c150fec0055c3e1c825c9503d6e2bcf638fcb57c33f71f46e8cb8b630","id":"ff3b16ca8ebc484fb57bcd75700dcb7e","object_ids":["20260710-065335-142a3abd","wired-evaluate-and-cli-entry-points"],"prev_hash":"9f2e75b6d37082ee754475878d52da597cc2cffaf42627b832adcf16a642df4c","reversible":true} diff --git a/.vouch/config.yaml b/.vouch/config.yaml index 108d471b..43c0ec1e 100644 --- a/.vouch/config.yaml +++ b/.vouch/config.yaml @@ -5,3 +5,5 @@ retrieval: backends: - fts5 - substring +compile: + llm_cmd: "claude -p --model sonnet" \ No newline at end of file diff --git a/CHANGELOG.md b/CHANGELOG.md index 986dcb0e..d50d67a3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,202 @@ All notable changes to vouch are documented here. Format follows ## [Unreleased] +## [1.3.0] — 2026-07-14 + +### Added +- `vouch console`: serve the vendored React web console straight from the + installed package — a same-origin `/proxy` bridge (loopback-guarded) to + `vouch serve --transport http` backends, reimplementing the vite dev-proxy + in python. the built SPA is bundled into the wheel as `vouch/web/console` + (conditionally, via a hatch build hook), so `pip install 'vouch-kb[web]'` + then `vouch console` needs no node and no repo clone. +- `kb.activity` read method (+ `vouch activity` CLI mirror): audit-log + activity buckets for dashboards — per-day counts with proposal/decision + breakdowns, an hour-of-week matrix, and actor/event histograms. windowed + in viewer-local calendar days (IANA `tz` or a fixed utc offset), scope- + filtered like `kb.audit`. +- console Dashboard view: 12-month activity calendar, last-30-days bars, + hour-of-week heatmap, top actors and event mix, driven by `kb.activity`. +- `kb.synthesize` llm backend: `llm=true` drafts the answer with the + deployment-configured `compile.llm_cmd`, grounded in retrieved kb pages + and approved claims. code still verifies every `[id]` citation against + the offered sources — invented ids are stripped, and a draft left with no + verifiable citation returns an empty answer rather than a guess. the wire + shape is unchanged plus additive `pages` and `_meta.synthesis_backend` + fields. cli mirror: `vouch synthesize --llm`; the jsonl/http surface + already forwarded the flag. +- console Chat: llm answers activated — the chat asks `kb.synthesize` with + `llm: true` and falls back to deterministic claim synthesis when no + `compile.llm_cmd` is configured. page citations open the page drawer, and + llm answers carry an `llm` badge next to the confidence grade. +- session transcript viewer: the console Review view opens the full + transcript of a captured session — thinking, per-tool call rendering + with diffs, code blocks, and subagent drill-down — via the new + `kb.session_transcript` read method. it locates the raw claude code + jsonl session file or codex rollout by id, parses it into normalized + blocks, and degrades to the observation buffer when the raw file is + gone. +- review-gated artifact delete: `kb.propose_delete` files a delete + proposal for a claim, page, entity, or relation. execution happens only + through `proposals.approve()`, which checks the referenced-by matrix, + removes the file, deindexes it, and records exactly what was removed in + the decided proposal and audit event. +- `kb.clear_claims` / `vouch claims-clear`: bulk-remove auto-approved + claims (`--auto-only`, `--before`, `--dry-run`, and a confirm gate) for + cleaning up capture noise in one pass (#433). +- session-split summaries: large captured sessions are split into topical + summary pages by a deployment-configured llm (`capture.split` in + `config.yaml`; the split `llm_cmd` falls back to `compile.llm_cmd`), + each filed as its own pending summary. `kb.summarize_session` runs the + pass on demand, `kb.list_sessions` lists captured sessions for the + review pipeline, and a filed mechanical summary can be llm-narrated in + place. +- codex adapter: `vouch install-mcp codex` now wires the full tier — a + `toml_merge` install strategy for `.codex/config.toml`, an agents.md + fenced snippet, skills mirroring the vouch slash commands, and hook + wiring for automatic session capture. codex session rollouts are + ingested into review-gated summaries. +- codex: wired `UserPromptSubmit` to `vouch context-hook` for the first + time, reusing the existing command unmodified — codex's hook + payload/response shape matches claude-code's exactly. (#425) +- `kb.list_skills` / `kb.get_skill` — agents can enumerate the Claude Code + slash-command and `SKILL.md` catalogue visible at `/.claude/` and + `~/.claude/` over MCP, then fetch the full body of one by name (project-local + entries override user-global on collision). Exposed across MCP (`kb_list_skills` + / `kb_get_skill`), JSONL, and the CLI (`vouch list-skills` / `vouch get-skill`). +- `mcp.publish_skills` config flag (default `true`) — gates the skill catalogue + for "company-brain" deployments where the catalogue itself is sensitive. When + `false`, `kb.list_skills` returns an empty list and `kb.get_skill` errors with + `permission_denied`; the flag is read fresh on every call so flipping it hides + the catalogue without restarting the server, and is surfaced on + `kb.capabilities.mcp.publish_skills` so clients can detect the gate. An + existing KB with no `mcp:` block stays default-on (#235). +- mcp serves a **minimal tool profile by default** (8 core tools) instead + of the full method surface; widen with `VOUCH_TOOL_PROFILE=standard|full` + or `mcp.tool_profile` in `config.yaml`. approve/reject and maintenance + tools live in `standard`/`full` — they stay human/cli actions, not agent + defaults. the jsonl and cli surfaces are unaffected. +- per-prompt auto-recall: the claude-code adapter's `UserPromptSubmit` + hook (`vouch context-hook`) injects relevant kb context on every + prompt, so recall no longer depends on the agent remembering to ask. +- `kb.experts` — rank the entities carrying the most matched evidence on + a free-text topic (count, recency, and citation weightings). read-only; + answers "who/what does this kb actually know about X" (#315). +- `kb.triage_pending` — advisory triage scoring over the pending-review + queue. scores each pending proposal on fit, citation quality, + duplication risk, and contradiction risk, then attaches a + `_meta.vouch_triage` block to help a reviewer prioritize a long + `kb.list_pending`. read-only and advisory only: it never approves, + rejects, or moves a proposal — a human still decides. degrades to a + `difflib` heuristic without the `[embeddings]` extra. opt-in via + `triage.enabled: true`; `vouch triage` mirrors it on the cli (#322). +- opt-in cross-encoder rerank of the context pack: enable with + `retrieval.rerank.enabled: true` in `config.yaml`; `retrieval.rerank.top_k` + bounds the rerank window. off by default (#436). +- `kb.diff` is registered at all four `kb.*` surface sites (mcp tool, + jsonl handler, capabilities, cli) instead of cli-only (#327). +- dual-solve web ui: file-changes tree view in the candidate panes — a compact + folders-first file tree drives a per-file diff pane, replacing the flat + changed-files list and the stacked all-files diff. selection is + per-candidate, so inspecting claude's diff never moves the codex pane. + (#294) +- demo: dual-path llm configuration — compile & summarize run through + session-capture replay or directly against the api via a stdlib shim + wired as `compile.llm_cmd` with a byo `ANTHROPIC_API_KEY`. + +### Changed +- retrieval `auto`/`hybrid` now **fuses embedding + fts5** results via + reciprocal rank fusion instead of a waterfall (embedding-first, + fts5-fallback), with near-duplicate suppression over the fused list — + the highest-scored near-duplicate wins, caller order is preserved for + the context pack. +- mcp serves one-line tool descriptions under non-full profiles, keeping + the default surface cheap in agent context windows. + +### Fixed +- audit: `log_event` serialises appends with a file lock, so concurrent + writers cannot fork the hash chain (#263). +- `lifecycle.contradict()` no longer lets a claim contradict itself. calling + it with the same claim id on both sides previously wrote a self-loop + `contradicts` reference and flipped the claim to `contested` with no + actual counterparty; it now raises `LifecycleError`, mirroring the + existing guard in `supersede()`. +- rpc internal errors no longer leak tracebacks over the wire; they log + server-side and return a clean error envelope. +- models reject empty `text`/`name`/`title` on claim, entity, and page at + validation time instead of filing empty artifacts. +- `kb.crystallize` retries are idempotent for summary pages — a re-run + after a partial failure no longer files a duplicate page proposal. +- `session.crystallize`: retrying on a session that hasn't been ended rewrote + the `session-` summary page with a fresh wall-clock `Ended:` stamp each + time (and needlessly re-embedded it), so the "idempotent retry" wasn't. an + open session now renders a stable marker, so retries produce an identical + body. +- claude-code: the `UserPromptSubmit` context hook computed retrieval but + never fed the entity-salience reflex (#223) — `salience.record_query` + was never called from the hook path, leaving the reflex permanently + dormant for every claude-code session. OpenClaw's context engine already + called it correctly; cursor's `beforeSubmitPrompt` hook cannot accept + injected context at all, so it is not wired. (#425) +- `vouch search` tolerates fts index errors instead of crashing — a + broken or stale fts table degrades to the substring path (#438). +- `list_pages` skips corrupt page files instead of failing the whole + listing, so one bad yaml no longer takes down every kb-wide caller + (#360). +- context: the `require_citations` gate is computed after the `max_chars` + budget is applied, so a claim trimmed out by the budget can no longer + satisfy (or fail) the citation requirement on the pack (#268). +- `vouch digest --limit` now caps the followups-due section like the + pending, decisions, and stale sections — it previously returned every + due followup regardless of the limit, contradicting the `--limit` help. +- the dual-solve diff renderer dropped added/removed lines whose content + starts with `++`/`--` (e.g. an added `++counter` line) by treating them as + `+++`/`---` file headers; the header skip now requires the trailing + space-and-path form. (#294) +- `compile_kb()` could file two page proposals for the same title when the + LLM's batch drafted the same topic (or same slug, e.g. "Retry Policy" vs + "retry policy") twice — `taken_names` was only seeded from on-disk pages + and pending proposals, never updated as drafts were accepted within the + batch. Approving the second proposal would silently route through + `update_page()` and overwrite the first. (#439) +- volunteer scoring treats hybrid relevance as rank-relative instead of + assuming pre-normalized scores. +- `kb.capabilities` reads `openclaw.compat` from `package.json` instead + of a stale manifest field (#417). +- the context hook never raises on a non-dict payload or a missing kb — + a broken hook environment degrades to no injection instead of failing + the host prompt. +- `context` and `lifecycle` catch only the exceptions they can handle + (sqlite errors on fts5 fallback, missing-artifact on citation lookup) + instead of blanket `except Exception`. +- `vouch install-mcp ` (codex `toml_merge`): a `config.toml` the minimal + serializer couldn't faithfully re-emit (a non-BMP string value, a `nan`/`inf` + float) was bucketed as `skipped` and printed as `(already present)` with a + clean `Done`, so the user believed vouch was wired into codex when it wasn't. + serializer-failure now lands in a distinct `failed` bucket, is reported as + such, and the command exits non-zero — "already installed" and "install + failed" no longer look the same. +- `vouch install-mcp `: a manifest `dst` that escaped the target tree + (via `..` or an absolute path) is now refused with an `AdapterError` instead + of writing outside `target` (defense in depth for the manifest file writer; + shipped adapters are unaffected). +- dual-solve review-ui: the recommendation hint rendered "neither engine + produced a usable diff" for the entire duration of a still-running job — the + hint was computed over the not-yet-populated candidate list on every poll. + it is now omitted until candidates exist. +- `vouch capture ingest-codex`: rollout parsing had no size cap and could read + an oversized (or newline-free blob) rollout whole into memory. the file is + now bounded to 64 MiB up front, mirroring the byte caps on other untrusted + reads. + +### Security +- `kb.register_source_from_path` blocks path traversal: the path is + resolved (following symlinks) and must land inside the kb root before + reading, with `O_NOFOLLOW` + `fstat` closing the toctou window between + the containment check and the read. previously any file the process + could access was registrable as a "source" and retrievable via + `kb.cite` / `kb.list_sources` (#421). + ## [1.2.2] — 2026-07-07 ### Packaging @@ -266,6 +462,11 @@ All notable changes to vouch are documented here. Format follows KB under `eval/fixture-kb/`, and an `eval` workflow gating retrieval changes (#226). ### Fixed +- `build_context_pack` now evaluates the `require_citations` gate (and + `quality.uncited_items`) after the `max_chars` budget drops tail items, so + the pack is never failed for uncited claims the caller did not receive. + Fixes #174. +- `audit.log_event` now holds an exclusive cross-process lock around read-prev-hash → derive → append, closing a TOCTOU race where two concurrent writers observed the same `prev_hash` and forked the chain — `verify_chain` then reported "previous hash mismatch" at the second concurrent event forever, breaking the tamper-evidence guarantee from #244 under ordinary multi-writer usage (`vouch serve` + concurrent CLI, multiple agents on JSONL, scripted backgrounded approvals). Uses `fcntl.flock` on POSIX and `msvcrt.locking` on Windows against a sibling `audit.log.jsonl.lock` file so the audit log itself is never opened in a mode that could truncate it. Fixes #262. - `parse_since` (the `--since` parser behind `vouch metrics`/`vouch audit`) now raises a clean `MetricsError` for a duration too large to represent (e.g. `--since 1000000000000d`), instead of letting an uncaught `OverflowError` traceback escape — restoring the documented "clean error, not a traceback" contract. - `sync_apply` now loads the sync source exactly once and passes the same `_SyncSource` instance into `sync_check`, closing a TOCTOU window where a bundle replaced on disk between the two `_load_source` calls could cause the validation and write phases to operate on different snapshots. Also eliminates redundant directory walks (KB sources) and triple tarball opens (bundle sources). Fixes #217. - `vault_to_kb` now passes `slug_hint=page_id` to `propose_page` so vault edit proposals target the existing page id from frontmatter instead of a slugified copy of the title (fixes #219). diff --git a/CLAUDE.md b/CLAUDE.md index 084a933d..49a472f9 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -69,7 +69,7 @@ Three rules that fall out of the layout: ```bash # from a clone python3 -m venv .venv && . .venv/bin/activate -pip install -e '.[dev]' +pip install -e '.[dev,web]' # dev,web is what ci.yml installs; mypy needs the web extra # the CI gate — exactly what .github/workflows/ci.yml runs .venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 7b06f924..33a4bf68 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -68,10 +68,15 @@ touch behavior. ```bash git clone https://github.com/vouchdev/vouch cd vouch -python -m venv .venv && source .venv/bin/activate -pip install -e '.[dev]' +python3 -m venv .venv && source .venv/bin/activate +pip install -e '.[dev,web]' ``` +The web console isn't prebuilt in a source checkout: `make webapp-build` +(needs node) builds it into `webapp/dist` so `vouch console` can serve +it, or `make console` runs the backend and the console dev server +together. + ## The gate `make check` is the same gate CI runs. The individual pieces: @@ -125,6 +130,9 @@ strict: - A `CHANGELOG.md` entry under `## [Unreleased]` for user-visible changes. - Lowercase prose in the PR body, matching the repo voice. No `Co-Authored-By` or AI-attribution trailers in commits. +- For changes that affect the webapp, CLI output, or user-facing behavior: + screenshots or screenrecording from the vouch-ui webapp showing the + behavior working as intended. ## Commit and PR titles diff --git a/Makefile b/Makefile index 021738a6..523e443f 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: help install dev test test-cov bench lint format type check build clean examples-screenshots smoke-capture smoke-recall +.PHONY: help install dev test test-cov bench lint format type check build clean examples-screenshots smoke-capture smoke-recall console webapp-build PY ?= python PIP ?= $(PY) -m pip @@ -19,6 +19,7 @@ help: @echo " make examples-screenshots re-render docs/img/examples/*.svg" @echo " make smoke-capture end-to-end check of session auto-capture" @echo " make smoke-recall end-to-end check of session-start recall" + @echo " make console run the vouch backend + vouch-ui web console together" install: $(PIP) install -e '.[dev]' @@ -58,9 +59,23 @@ smoke-capture: smoke-recall: VOUCH="$(PY) -m vouch" bash scripts/smoke-recall.sh -build: +# Run the vouch HTTP backend and the vouch-ui web console together (Ctrl-C stops +# both). Installs the web console's node deps on first run. +console: + bash scripts/console.sh + +# Build the React console into webapp/dist so the wheel can bundle it as +# vouch/web/console (hatch_build.py force-includes it when present). Installs +# the web console's node deps on first run. +webapp-build: + cd webapp && { [ -d node_modules ] || npm ci; } && npm run build + +# sdist is source-only; the wheel is built from the working tree (not the +# sdist) so the freshly-built, gitignored webapp/dist rides inside it. +build: webapp-build $(PY) -m pip install --upgrade build - $(PY) -m build + $(PY) -m build --sdist + $(PY) -m build --wheel clean: rm -rf build dist *.egg-info src/*.egg-info \ diff --git a/README.md b/README.md index 0e9506bb..39056961 100644 --- a/README.md +++ b/README.md @@ -32,6 +32,27 @@ Everything below exists to reproduce that loop on your own project. ## Install +**For the full UI experience** (recommended first time): + +```bash +docker run --rm -p 127.0.0.1:5173:5173 -v vouch-demo-data:/data ghcr.io/plind-junior/vouch-demo +# then open http://localhost:5173 +``` + +Pre-seeded KB + full webapp console, zero setup. Pass `-e ANTHROPIC_API_KEY=sk-ant-...` to enable LLM features. + +**For the full UI without Docker** — Python only, no clone, no node: + +```bash +pipx install 'vouch-kb[web]' # the browser console ships inside the wheel +vouch serve --transport http --port 8731 & # a backend for the current .vouch/ +vouch console # console at http://localhost:5173 — connect it to :8731 +``` + +`vouch console` serves the same React console as the Docker demo, straight from the installed package. + +**For CLI + Claude Code integration** (most common ongoing workflow): + ```bash # one-liner (Linux + macOS) — picks a Python, ensures pipx, installs vouch-kb curl -fsSL https://raw.githubusercontent.com/vouchdev/vouch/main/install.sh | sh @@ -40,24 +61,45 @@ curl -fsSL https://raw.githubusercontent.com/vouchdev/vouch/main/install.sh | sh pipx install vouch-kb ``` -The one-liner is POSIX `sh` and never needs `sudo` — inspect [`install.sh`](install.sh) first if you'd like. Prefer containers? The released image runs the same CLI and MCP server ([`ghcr.io/vouchdev/vouch`](https://github.com/vouchdev/vouch/pkgs/container/vouch)): +The one-liner is POSIX `sh` and never needs `sudo` — inspect [`install.sh`](install.sh) first if you'd like. + +**For MCP server or CLI-only use**: ```bash docker run -i --rm -v "$PWD:/data" ghcr.io/vouchdev/vouch:latest # stdio MCP server docker run --rm -v "$PWD:/data" ghcr.io/vouchdev/vouch:latest status # any CLI command ``` +**For local development** — CLI and webapp, both running from source: + +```bash +git clone https://github.com/vouchdev/vouch +cd vouch +python3 -m venv .venv && source .venv/bin/activate +pip install -e '.[dev,web]' # dev,web is what CI installs — make check needs both +vouch --version # the CLI now runs straight from src/ — edits apply without reinstalling + +make console # webapp in dev mode: vouch backend on :8731 + live-reload + # console at http://localhost:5173 — Ctrl-C stops both + +make check # the CI gate: lint + type + test +``` + +`make console` needs node — it starts `vouch serve --transport http` and the Vite dev server as a pair, installing the console's node deps automatically on first run. To instead serve the console the way a release wheel does (no dev server), run `make webapp-build` once, then `vouch console`. See [CONTRIBUTING.md](CONTRIBUTING.md) for the full dev workflow. + ## Reproduce the loop on your project +After exploring the demo above, set up vouch in your own project: + **1. Set up the KB and wire Claude Code** (one-time, per repo): ```bash cd /path/to/your/project -vouch init -vouch install-mcp claude-code +vouch init # creates .vouch/ with starter config +vouch install-mcp claude-code # wires capture hooks into Claude Code ``` -`init` creates `.vouch/` with a starter config; `install-mcp` writes `.mcp.json` (the `kb.*` MCP tools), the `/vouch-*` slash commands, and three hooks — `PostToolUse` capture, `SessionEnd` rollup, `SessionStart` recall. Restart Claude Code so they load. +`install-mcp` writes `.mcp.json` (the `kb.*` MCP tools), the `/vouch-*` slash commands, and three hooks — `PostToolUse` capture, `SessionEnd` rollup, `SessionStart` recall. Restart Claude Code so they load. **2. Point `compile` at an LLM** — the only step that needs a model. In `.vouch/config.yaml`: @@ -78,16 +120,43 @@ compile: vouch review # walk pending proposals one at a time ``` -The browser console in the video is the **[vouch webapp](https://github.com/vouchdev/webApp)** — chat, review, pending queue, claims, and stats over a running KB. Connect it in two commands: +**Want a browser UI for reviewing and proposing?** The video shows the **vouch webapp** — chat, review queue, claims, and stats. You have four options: + +- **No setup**: Use the Docker demo (recommended) +- **pip, no clone**: `pipx install 'vouch-kb[web]'` then `vouch console` — serves the same React console from the installed package (Python only, no Docker, no node), open http://localhost:5173 +- **Local development**: Clone the repo, run `make console`, open http://localhost:5173 +- **CLI-only**: Use `vouch review`, `vouch show `, `vouch approve ` commands instead + +**Point the webapp at your existing KB:** + +```bash +# Terminal 1: start the vouch server pointing at your .vouch/ +cd /path/to/your/project +vouch serve --transport http --port 8731 + +# Terminal 2: run the Docker UI pointing at that server +docker run --rm -p 127.0.0.1:5173:5173 \ + -e VOUCH_TARGET=http://host.docker.internal:8731 \ + ghcr.io/plind-junior/vouch-demo +# then open http://localhost:5173 +``` + +Or serve that same console with no Docker — `vouch console` in place of Terminal 2 (needs the `[web]` extra), then add the `:8731` backend in the connect dialog: + +```bash +vouch console # http://localhost:5173, proxying to the server above +``` + +Or to skip the browser entirely and use the CLI tools: ```bash -vouch serve --transport http # serves the kb.* surface on 127.0.0.1:8731 -# then, in a clone of the vouch webapp: -npm install && npm run dev # opens http://localhost:5173 — point the - # connect dialog at http://127.0.0.1:8731 +vouch review # walk pending proposals +vouch show # inspect a claim or page +vouch approve # approve a proposal +vouch reject --reason "…" # reject with feedback ``` -Lighter alternatives ship with vouch itself: `vouch review-ui` (a built-in browser queue; `pipx install 'vouch-kb[web]'` for the extra), or piecemeal `vouch pending`, `vouch show `, `vouch approve `, `vouch reject --reason "…"`. +Both browser UIs ship with vouch under the `[web]` extra (`pipx install 'vouch-kb[web]'`): `vouch console` is the full React console shown in the video; `vouch review-ui` is a lighter built-in review queue. Or go piecemeal: `vouch pending`, `vouch show `, `vouch approve `, `vouch reject --reason "…"`. **5. Compile the wiki.** diff --git a/ROADMAP.md b/ROADMAP.md index 46101d1f..78110c57 100644 --- a/ROADMAP.md +++ b/ROADMAP.md @@ -4,62 +4,82 @@ This is a rough plan, not a contract. Anything dated is a target, not a promise. Items marked **[VEP]** require a written proposal in [proposals/](proposals/) before implementation. -## 0.1 — surface stabilises (next) - -- Vector embeddings as a retrieval backend alongside FTS5 (shipped). - Retrieval is controlled by `retrieval.backend` in `config.yaml`: - `auto` (default) tries embedding → FTS5 → substring, gracefully - degrading to FTS5 when the embeddings extras aren't installed; set - `embedding`, `fts5`, or `substring` to pin a single path. -- `vouch diff ` for claim/page revisions. -- `vouch approve --batch` for reviewing N proposals in one transaction. -- HTTP transport (`vouch serve --transport http`) behind a localhost - bind by default. **[VEP]** -- Migration story: `vouch migrate` to upgrade on-disk layout between - minor versions without losing the audit trail. - -## 0.2 — multi-agent & scopes - -- Scopes on Claim and Source beyond a single field — at minimum - `(visibility, project, agent)` so a multi-agent KB can carve up - who-sees-what. **[VEP]** -- Multi-agent sync: well-defined merge semantics for two `.vouch/` - directories that diverged. Today this is "git merge and hope"; - we want a deterministic resolver for `decided/` and `audit.log.jsonl`. -- Adapter templates checked in for the major runtimes - (Claude Code, Cursor, Codex, Continue). -- Conformance suite: a runnable test pack that any KB server claiming - to speak `kb.*` can be measured against. - -## 0.3 — operational maturity - -- Benchmarks: search latency, proposal throughput, bundle import time - on KBs in the 1k / 10k / 100k claim range. +The 0.x milestones that used to live here shipped in the 1.x line: +embeddings + hybrid retrieval, `vouch diff`, HTTP transport, migrations, +adapter templates for the major runtimes, and the web console. What +follows is what's next. + +## 1.3.x — integrity & retrieval defaults + +- Audit-log hardening: file locking on the hash-chain append, and + audit-before-move ordering in `approve()` so a crash can never leave a + durable claim without its audit event. +- Full three-surface parity enforcement: the capabilities test compares + MCP tools and JSONL handlers today; the CLI mirror joins it. +- Zero-config retrieval quality: hybrid fusion is already the default; + next is rerank + recency in the default read path (not flag-only) and + a story for local embeddings that works on a base install. +- Retrieval honesty: when the backend degrades (no embeddings, FTS5 + unavailable), say so in `_meta` instead of silently returning worse + results. + +## 1.4 — the wiki front door + +Pages are the product; claims are the citation layer. This milestone +makes the wiki browsable and self-maintaining: + +- A `/page` browse surface in the console (the chat's page drawer grows + into a real reader with an index). +- Backlinks: `[[wikilinks]]` are validated today, then discarded; + persist the link graph and render it. +- Wiki lint: orphan pages, stale pages, dead links, uncited sections. +- `vouch compile` maintains existing pages (compounding updates), not + just new ones. +- Good synthesized answers can be filed back as page proposals — gated + by review like every other write. + +## 1.5 — measurement + +- A committed, reproducible eval harness beyond the recall gate: + follow-rate tracking with confidence intervals and a coding-specific + corpus. +- Benchmarks: search latency, proposal throughput, bundle import time on + KBs in the 1k / 10k / 100k claim range. - `vouch fsck` — deeper consistency checks than `doctor`. -- Structured logging behind `VOUCH_LOG_FORMAT=json`. -- First-class observability hooks (proposal counts, approval rate, - citation-coverage). -## 1.0 — frozen on-disk format + frozen method surface +## 2.0 — connected KBs + +Multiple vouch instances — personal, per-project, team — exchanging +reviewed knowledge. The review gate is non-negotiable: a receiving KB +accepts inbound knowledge as *proposals*; nothing lands past +`proposals.approve()`. Self-hosted, local-first — not a hosted service. -Once we cut 1.0: -- The on-disk layout in `.vouch/` is a stable format. Breaking changes - require a major bump and a migration tool. -- The `kb.*` method surface is stable. Adding methods is fine; removing - or changing signatures requires a major bump. -- Semantic versioning applies normally from this point. +- Deterministic merge semantics for two `.vouch/` directories that + diverged (replaces "git merge and hope") — the audit hash chain is + what makes divergence detectable. **[VEP]** +- Bundle push/pull between KBs: signed bundles of decided claims and + pages that arrive as proposals, actor identity preserved end-to-end in + the audit log. **[VEP]** +- Scopes beyond a single field — at minimum `(visibility, project, + agent)` so a multi-KB deployment can carve up who-sees-what. **[VEP]** +- A hub daemon: registry of connected KBs, scope-based subscription + rules, federated search with provenance (every result names the KB + that vouched for it), and a hub view in the console. +- Conformance suite: a runnable test pack that any KB server claiming to + speak `kb.*` can be measured against — connecting is a protocol + property, not a product feature. ## Explicitly out of scope (today) These come up in discussion. The current answer is "not now": -- **Hosted vouch service.** vouch is a library + CLI. A hosted - multi-tenant version is somebody else's product. -- **A web UI for review.** PRs are the review UI. A standalone web - reviewer might be nice but it isn't on this list. +- **Hosted vouch service.** vouch is a library + CLI + self-hosted + console. A hosted multi-tenant version is somebody else's product. +- **Removing the review gate for "trusted" agents.** The gate is the + product. - **Cross-language clients beyond the protocol.** If you want a - TypeScript client, implement the `kb.*` JSONL contract; we'll keep - it documented. + TypeScript client, implement the `kb.*` JSONL contract; we keep it + documented and (in 2.0) conformance-testable. If something here matters to you, please file an issue — order is negotiable. diff --git a/adapters/codex/hooks.json b/adapters/codex/hooks.json index acaf8d04..d110fd74 100644 --- a/adapters/codex/hooks.json +++ b/adapters/codex/hooks.json @@ -1,5 +1,16 @@ { "hooks": { + "UserPromptSubmit": [ + { + "hooks": [ + { + "type": "command", + "command": "vouch context-hook", + "statusMessage": "vouch context" + } + ] + } + ], "Stop": [ { "hooks": [ diff --git a/adapters/codex/install.yaml b/adapters/codex/install.yaml index 9d4cb447..ea678f1f 100644 --- a/adapters/codex/install.yaml +++ b/adapters/codex/install.yaml @@ -38,14 +38,19 @@ tiers: - { src: ../openclaw/skills/vouch-record/SKILL.md, dst: .codex/skills/vouch-record/SKILL.md } - { src: ../openclaw/skills/vouch-followup/SKILL.md, dst: .codex/skills/vouch-followup/SKILL.md } - { src: ../openclaw/skills/vouch-standup/SKILL.md, dst: .codex/skills/vouch-standup/SKILL.md } - # T4 = automatic session capture -- see issue #388. codex's hooks system - # fires Stop when a turn completes; the handler re-ingests the session's - # rollout idempotently (`vouch capture ingest-codex --hook` exits 0 even - # on failure, so capture can never break a codex turn), updating the - # session's single PENDING summary proposal as the session grows. - # hooks live in their own `.codex/hooks.json` file (project-local in - # trusted projects) and json_merge preserves any hooks the user already - # has. note: the legacy `notify` setting can't be used here -- codex only - # honours it in user-global config, which the #179 rule forbids touching. + # T4 = automatic session capture (issue #388) plus per-prompt KB context + # injection (issue #425). codex's hooks system fires Stop when a turn + # completes; the handler re-ingests the session's rollout idempotently + # (`vouch capture ingest-codex --hook` exits 0 even on failure, so capture + # can never break a codex turn), updating the session's single PENDING + # summary proposal as the session grows. codex also fires UserPromptSubmit + # with the same {prompt, session_id, ...} shape and additionalContext + # response contract as claude-code's UserPromptSubmit, so the same + # `vouch context-hook` command (see hooks.py, #425) serves both hosts + # unmodified. hooks live in their own `.codex/hooks.json` file + # (project-local in trusted projects) and json_merge preserves any hooks + # the user already has. note: the legacy `notify` setting can't be used + # here -- codex only honours it in user-global config, which the #179 + # rule forbids touching. T4: - { src: hooks.json, dst: .codex/hooks.json, json_merge: true } diff --git a/adapters/cursor/install.yaml b/adapters/cursor/install.yaml index 900a752f..e75f1e4e 100644 --- a/adapters/cursor/install.yaml +++ b/adapters/cursor/install.yaml @@ -5,6 +5,15 @@ # `~/.cursor/mcp.json` is intentionally out of scope (see #179 acceptance — # we don't touch user-global config from a project-scoped install). # T2 = AGENTS.md fenced snippet (Cursor reads AGENTS.md the way Claude Code reads CLAUDE.md). +# +# Scope decision (issue #425): vouch's per-prompt KB-context hook was NOT +# mirrored to Cursor. Cursor's closest analogue, beforeSubmitPrompt, is +# validation/block-only -- it can inspect and reject a prompt but cannot +# inject additionalContext the way claude-code's and codex's +# UserPromptSubmit can. Cursor's own sessionStart hook does support +# additional_context, but that fires once per session, not per prompt, so +# it isn't a substitute for the reflex-driven per-turn injection #425 asks +# for. Revisit if Cursor ships prompt-time context injection. host: cursor pretty: Cursor fence: diff --git a/adapters/openclaw/install.yaml b/adapters/openclaw/install.yaml index efd2ff8a..49dade1c 100644 --- a/adapters/openclaw/install.yaml +++ b/adapters/openclaw/install.yaml @@ -18,6 +18,16 @@ # from adapters/claude-code/ directly). # T4 = `.openclaw/policy.json` -- the trust boundary as project-local # policy (review-gated writes, audit-logged lifecycle, confined fs). +# +# Note re issue #425: OpenClaw needs no changes here. Its per-prompt context +# injection isn't a hooks.json-style shell hook at all -- it's the +# context-engine slot (openclaw.plugin.json's kind: "context-engine", +# adapters/openclaw/vouch-context-engine.mjs bridging to +# src/vouch/openclaw/context_engine.py's assemble()), and that engine +# already calls salience.record_query / attach_salience on every assemble() +# (see src/vouch/openclaw/context_engine.py). #425's fix was specifically +# for the claude-code/codex hooks.py path, which had the reflex dead; +# OpenClaw's own path never had that bug. host: openclaw pretty: OpenClaw fence: diff --git a/demo/.env.example b/demo/.env.example new file mode 100644 index 00000000..f4d675ba --- /dev/null +++ b/demo/.env.example @@ -0,0 +1,18 @@ +# vouch demo — environment (optional). Copy to .env to override defaults. +# +# cp .env.example .env + +# Bearer token vouch requires on its HTTP transport. The console injects it +# server-side, so the browser never sees it. Any non-empty value works for the +# local demo; change it if you like. +VOUCH_HTTP_TOKEN=vouch-demo + +# Optional — your own Anthropic API key. Set it to turn on the two LLM-backed +# actions in the console: "Compile" (approved claims -> topic pages) and +# "Summarize session". Leave it blank and the demo still runs; those two just +# report "not configured". The key stays in this container, calls go straight +# to Anthropic, and nothing is committed. +ANTHROPIC_API_KEY= + +# Optional — override the model the shim uses (defaults to a current Sonnet). +# ANTHROPIC_MODEL=claude-sonnet-4-5 diff --git a/demo/Dockerfile b/demo/Dockerfile new file mode 100644 index 00000000..df498de4 --- /dev/null +++ b/demo/Dockerfile @@ -0,0 +1,59 @@ +# syntax=docker/dockerfile:1 +# +# vouch demo — one image, one command. Bundles the vouch server (built from +# THIS checkout, so it carries the newest kb.* surface incl. delete / archive / +# supersede), the vouch-ui web console, and seeds a starter knowledge base on +# first run. `docker compose up` in this folder, open the browser, and explore +# a populated, review-gated KB. +# +# Two runtimes in one image on purpose: python runs `vouch serve`, node serves +# the console via `vite preview` (which reuses the console's own /proxy/* +# middleware, pinned at the in-container vouch endpoint). + +# ---- stage 1: build the console to static (keep the tree for vite preview) --- +FROM node:22-slim AS web +WORKDIR /web +COPY webapp/package.json webapp/package-lock.json ./ +RUN npm ci +COPY webapp/ ./ +RUN npm run build + +# ---- stage 2: runtime = node (console) + python (vouch from source) --------- +FROM node:22-slim +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + LANG=C.UTF-8 \ + NODE_ENV=production \ + VOUCH_UI_ALLOW_REMOTE=1 \ + VOUCH_TARGET=http://127.0.0.1:8731 \ + VOUCH_HTTP_TOKEN=vouch-demo \ + VOUCH_DATA_DIR=/data \ + ANTHROPIC_MODEL=claude-sonnet-4-5 + +RUN apt-get update && apt-get install -y --no-install-recommends \ + python3 python3-venv curl \ + && rm -rf /var/lib/apt/lists/* + +# vouch from this checkout — the [web] extra brings the HTTP transport in. +COPY pyproject.toml README.md /src/ +COPY src /src/src +COPY adapters /src/adapters +RUN python3 -m venv /opt/venv && /opt/venv/bin/pip install --no-cache-dir "/src[web]" +ENV PATH="/opt/venv/bin:$PATH" + +# the built console tree (vite preview needs vite + plugins + dist/ at runtime) +COPY --from=web /web /app/webapp + +# bring-your-own-key LLM shim: reads a prompt on stdin, calls the Anthropic +# Messages API from ANTHROPIC_API_KEY. Wired in as compile.llm_cmd by the +# entrypoint only when a key is present. Stdlib only — runs on the venv python. +COPY demo/vouch-llm.py /usr/local/bin/vouch-llm +RUN chmod +x /usr/local/bin/vouch-llm + + +COPY demo/entrypoint.sh /entrypoint.sh +RUN chmod +x /entrypoint.sh + +VOLUME ["/data"] +EXPOSE 5173 +ENTRYPOINT ["/entrypoint.sh"] diff --git a/demo/README.md b/demo/README.md new file mode 100644 index 00000000..e47eaaa7 --- /dev/null +++ b/demo/README.md @@ -0,0 +1,121 @@ +# vouch demo — try it in one command + +A self-contained Docker demo of [vouch](https://github.com/vouchdev/vouch), the +git-native, **review-gated** knowledge base for LLM agents. One image bundles +the vouch server and the vouch-ui web console, and seeds a starter knowledge +base on first run — so you open the browser and immediately have something to +explore. + +## Run it (no clone needed) + +One command pulls the published image and starts everything: + +```bash +docker run --rm -p 127.0.0.1:5173:5173 -v vouch-demo-data:/data \ + ghcr.io/plind-junior/vouch-demo +``` + +Then open **http://localhost:5173** and connect with the pre-filled endpoint. + +That's it. The first run seeds a starter KB (a claim, a page, a source), so the +console opens onto a populated, review-gated knowledge base — not an empty one. +Your data persists in the `vouch-demo-data` volume between runs; `Ctrl-C` stops +the demo (the `--rm` removes only the container, never the volume). + +## Update to the latest + +The image is updated in place, so pulling gets you the newest build (and any +fixes). Stop the demo, then: + +```bash +docker pull ghcr.io/plind-junior/vouch-demo # fetch the latest image +``` + +Re-run the command from "Run it" — Docker now starts the updated image, and +your `vouch-demo-data` volume carries over. To start completely fresh instead, +reset the data with `docker volume rm vouch-demo-data` before running. + +## Build from source (to hack on it) + +Working in a clone of this repo? Build the image from the checkout instead of +pulling — it picks up your local changes to `webapp/` and `src/`: + +```bash +cd demo +cp .env.example .env # optional: edit VOUCH_HTTP_TOKEN, add ANTHROPIC_API_KEY +docker compose up --build +``` + +## Turn on the LLM features (optional) + +Two console actions call a language model: **Compile** (turn approved claims +into topic pages) and **Summarize session**. They're off by default because +the demo ships no API key. Everything else — browsing, propose → approve, +delete / archive / supersede, clear queue — works without one. + +To switch them on, give the demo *your own* Anthropic key. With the pulled +image, pass it in with `-e`: + +```bash +docker run --rm -p 127.0.0.1:5173:5173 -v vouch-demo-data:/data \ + -e ANTHROPIC_API_KEY=sk-ant-... \ + ghcr.io/plind-junior/vouch-demo +``` + +Building from source? Put it in `.env` instead: + +```bash +cp .env.example .env # then set ANTHROPIC_API_KEY=sk-ant-... +docker compose up --build +``` + +The key never reaches the browser — vouch runs a tiny stdlib shim +(`vouch-llm`) inside the container that calls the Anthropic Messages API +directly. Override the model with `ANTHROPIC_MODEL` if you want a newer Sonnet. +Without a key, Compile / Summarize simply report "not configured" — that's the +review gate telling you the step is unavailable, not a crash. + +## What you can do in the console + +- **Browse / Claims** — the seeded knowledge, with citations and provenance + ("why does this claim exist?"). +- **Pending** — the propose → approve review gate, made visible. +- **Delete / Archive / Supersede** — retire a claim through the gate: delete + files a review-gated proposal (refused if other pages still cite it, which is + the point), archive hides it from retrieval, supersede replaces it. +- **Clear queue** — reject the whole pending queue at once. + +## How it works + +One container runs two processes (managed by `entrypoint.sh`): + +1. `vouch serve --transport http` on `127.0.0.1:8731` inside the container, with + a bearer token. +2. the console via `vite preview`, whose `/proxy/*` middleware forwards to the + in-container vouch endpoint. The token is **pinned and injected server-side** + (`VOUCH_TARGET` / `VOUCH_HTTP_TOKEN`), so the browser never sees it. + +Only the console port is published, and only on `127.0.0.1` — nothing leaves +your machine. Your KB lives in the `vouch-demo-data` volume: + +```bash +# pulled image (docker run): +Ctrl-C # stop (keeps your data) +docker volume rm vouch-demo-data # reset the demo KB + +# built from source (docker compose): +docker compose down # stop (keeps your data) +docker compose down -v # stop and reset the demo KB +``` + +## Notes + +- The published image (`ghcr.io/plind-junior/vouch-demo`) carries the newest + `kb.*` surface, including delete / archive / supersede — a released `vouch` + from PyPI or `ghcr.io/vouchdev/vouch` would not yet advertise those. Building + from source picks up whatever is in your checkout. +- The LLM-backed actions (Compile, Summarize session) need an + `ANTHROPIC_API_KEY` — see "Turn on the LLM features" above. The rest of the + console is fully functional without one. +- The console's "Chat / Claude Code" mode is a dev-server feature and is not + wired up in this preview build; everything else works. diff --git a/demo/docker-compose.yml b/demo/docker-compose.yml new file mode 100644 index 00000000..8b116b13 --- /dev/null +++ b/demo/docker-compose.yml @@ -0,0 +1,39 @@ +# Try vouch in one command: +# +# cd demo +# docker compose up --build # then open http://localhost:5173 +# +# A starter, review-gated knowledge base is seeded on first run. Your data +# persists in the `vouch-demo-data` volume; `docker compose down -v` resets it. +# Only the console port is published, and only on 127.0.0.1 — nothing leaves +# your machine. +name: vouch-demo + +services: + vouch-demo: + build: + context: .. + dockerfile: demo/Dockerfile + image: vouch-demo:latest + ports: + - "127.0.0.1:5173:5173" + environment: + # in-container bearer token; injected server-side, never sent to the browser + VOUCH_HTTP_TOKEN: ${VOUCH_HTTP_TOKEN:-vouch-demo} + # optional: your own Anthropic key turns on page compile & session + # summaries (Claude). Left empty, the demo still runs — those two actions + # just report "not configured". Never commit a real key. + ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY:-} + ANTHROPIC_MODEL: ${ANTHROPIC_MODEL:-claude-sonnet-4-5} + volumes: + - vouch-demo-data:/data + healthcheck: + test: ["CMD", "curl", "-fsS", "-m", "3", "http://127.0.0.1:5173/proxy/health"] + interval: 30s + timeout: 5s + retries: 3 + start_period: 20s + restart: unless-stopped + +volumes: + vouch-demo-data: diff --git a/demo/entrypoint.sh b/demo/entrypoint.sh new file mode 100644 index 00000000..a3d6869a --- /dev/null +++ b/demo/entrypoint.sh @@ -0,0 +1,78 @@ +#!/usr/bin/env bash +# +# vouch demo entrypoint: seed a starter KB on first run, start the vouch server +# on loopback inside this container, then serve the console. Both processes live +# in one container; killing the container stops both. +set -euo pipefail + +DATA="${VOUCH_DATA_DIR:-/data}" +TOKEN="${VOUCH_HTTP_TOKEN:-vouch-demo}" +# actor recorded in the seeded KB's audit log (avoids getpass in a bare image) +export VOUCH_USER="${VOUCH_USER:-demo}" + +# First run: an empty /data volume gets a seeded, review-gated starter KB so the +# console has something to show. Idempotent — skipped once .vouch/ exists. +if [ ! -d "$DATA/.vouch" ]; then + echo "[demo] seeding a starter knowledge base in $DATA ..." + vouch init --path "$DATA" +fi + +# vouch's LLM features (page compile, session summaries) shell out to the +# command in `compile.llm_cmd`. Support two workflows: +# 1. If Claude CLI is available (~/.claude exists), use `claude -p` to capture +# in Claude sessions (same as the real vouch project). +# 2. If only ANTHROPIC_API_KEY is set, use the direct API shim (vouch-llm). +# 3. If neither, leave LLM features unset so actions return "not configured". +CONFIG="$DATA/.vouch/config.yaml" +LLM_CMD="" +if [ -d "$HOME/.claude" ]; then + LLM_CMD="claude -p --model sonnet-4-5" + echo "[demo] Claude CLI found — LLM features wire to 'claude -p', compile & summarize will capture in Claude sessions." +elif [ -n "${ANTHROPIC_API_KEY:-}" ]; then + LLM_CMD="vouch-llm" + echo "[demo] ANTHROPIC_API_KEY set — LLM features enabled via direct API (compile & summarize will NOT capture in Claude sessions; use 'claude login' + mount ~/.claude to enable session capture)." +else + echo "[demo] LLM features DISABLED — provide ANTHROPIC_API_KEY or mount ~/.claude (with 'claude login' done) to enable compile & summarize." +fi + +if [ -n "$LLM_CMD" ]; then + python3 - "$CONFIG" "$LLM_CMD" <<'PY' +import sys, yaml +path, cmd = sys.argv[1:3] +with open(path) as f: + cfg = yaml.safe_load(f) or {} +cfg.setdefault("compile", {})["llm_cmd"] = cmd +with open(path, "w") as f: + yaml.safe_dump(cfg, f, sort_keys=False) +PY +else + python3 - "$CONFIG" <<'PY' +import sys, yaml +path = sys.argv[1] +with open(path) as f: + cfg = yaml.safe_load(f) or {} +if isinstance(cfg.get("compile"), dict): + cfg["compile"].pop("llm_cmd", None) + if not cfg["compile"]: + cfg.pop("compile") +with open(path, "w") as f: + yaml.safe_dump(cfg, f, sort_keys=False) +PY +fi + +# vouch on loopback (same container). A token is set so the console's proxy can +# inject it server-side; the browser never sees it. +echo "[demo] starting vouch server on 127.0.0.1:8731 ..." +( cd "$DATA" && exec vouch serve --transport http --host 127.0.0.1 --port 8731 --token "$TOKEN" ) & +VOUCH_PID=$! +trap 'kill "$VOUCH_PID" 2>/dev/null || true' EXIT INT TERM + +# Wait for vouch to answer its public liveness probe before starting the UI. +for _ in $(seq 1 30); do + if curl -fsS -m 2 "http://127.0.0.1:8731/health" >/dev/null 2>&1; then break; fi + sleep 1 +done + +echo "[demo] starting the vouch console on :5173 — open http://localhost:5173" +cd /app/webapp +exec npm run preview -- --host 0.0.0.0 --port 5173 diff --git a/demo/vouch-llm.py b/demo/vouch-llm.py new file mode 100644 index 00000000..ead3a180 --- /dev/null +++ b/demo/vouch-llm.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""Minimal Anthropic Messages shim for the vouch demo image. + +vouch's LLM-backed features (page compile, session summaries) don't call an +API directly — they shell out to a deployment-configured command +(`compile.llm_cmd` in .vouch/config.yaml) with the prompt on stdin, and read +the model's reply from stdout. In a normal install that command is the local +`claude` CLI. The demo image has no CLI and no baked-in key, so this shim is +the `llm_cmd`: it reads the prompt on stdin and calls the Anthropic Messages +API using a key the *user* supplies via ANTHROPIC_API_KEY. + +Stdlib only (urllib) — no extra pip dependency, mirroring vouch's own client +in src/vouch/pr_cache.py. Emits only the model's text on stdout so vouch's +`parse_drafts` sees a clean JSON array; all diagnostics go to stderr, and a +non-zero exit lets vouch surface a clean "compile.llm_cmd failed" message. + +Env: + ANTHROPIC_API_KEY required — user's key; absent => exit 3, features off. + ANTHROPIC_MODEL default claude-sonnet-4-5 (override for a newer Sonnet). + ANTHROPIC_BASE_URL default https://api.anthropic.com + ANTHROPIC_MAX_TOKENS default 8192 (compile/split return multi-page JSON). + ANTHROPIC_TIMEOUT default 150 (seconds; below vouch's own subprocess cap). +""" +from __future__ import annotations + +import json +import os +import sys +import urllib.error +import urllib.request + + +def main() -> int: + key = os.environ.get("ANTHROPIC_API_KEY", "").strip() + if not key: + sys.stderr.write( + "ANTHROPIC_API_KEY is not set — this demo's LLM features " + "(page compile, session summaries) are disabled. Set the key and " + "restart to enable Claude.\n" + ) + return 3 + + model = os.environ.get("ANTHROPIC_MODEL", "claude-sonnet-4-5").strip() + base = os.environ.get("ANTHROPIC_BASE_URL", "https://api.anthropic.com").rstrip("/") + try: + max_tokens = int(os.environ.get("ANTHROPIC_MAX_TOKENS", "8192")) + timeout = float(os.environ.get("ANTHROPIC_TIMEOUT", "150")) + except ValueError as e: + sys.stderr.write(f"invalid ANTHROPIC_MAX_TOKENS/ANTHROPIC_TIMEOUT: {e}\n") + return 2 + + prompt = sys.stdin.read() + payload = json.dumps({ + "model": model, + "max_tokens": max_tokens, + "messages": [{"role": "user", "content": prompt}], + }).encode() + req = urllib.request.Request( + f"{base}/v1/messages", + data=payload, + method="POST", + headers={ + "content-type": "application/json", + "x-api-key": key, + "anthropic-version": "2023-06-01", + }, + ) + + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + body = resp.read() + except urllib.error.HTTPError as e: + detail = (e.read().decode("utf-8", "replace") or "").strip()[:400] + sys.stderr.write(f"anthropic API {e.code}: {detail}\n") + return 1 + except (urllib.error.URLError, TimeoutError) as e: + sys.stderr.write(f"anthropic API call failed: {e}\n") + return 1 + + try: + data = json.loads(body) + except json.JSONDecodeError: + sys.stderr.write(f"anthropic API returned non-JSON: {body[:200]!r}\n") + return 1 + + text = "".join( + block.get("text", "") + for block in (data.get("content") or []) + if isinstance(block, dict) and block.get("type") == "text" + ) + if not text.strip(): + sys.stderr.write(f"anthropic API returned no text content: {body[:200]!r}\n") + return 1 + + sys.stdout.write(text) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/docs/INSTALL_FOR_AGENTS.md b/docs/INSTALL_FOR_AGENTS.md new file mode 100644 index 00000000..3ccb2ff9 --- /dev/null +++ b/docs/INSTALL_FOR_AGENTS.md @@ -0,0 +1,151 @@ +# install for agents + +machine-readable, top-to-bottom checklist for an assistant dropped into a +fresh project. every step is a concrete `vouch` command plus an assertion on +its output. if a step fails, stop and fix it before continuing. + +this doc is for agents. humans should start with +[getting-started.md](getting-started.md). + +## prerequisites + +1. confirm `vouch` is on `PATH`: + + ```bash + vouch --version + ``` + + **assert:** prints a version string and exits 0. + +2. `cd` to the project root you are wiring (the tree that will hold `.vouch/`). + +## 1. detect the host + +```bash +vouch install-mcp --list +``` + +**assert:** + +- exits 0. +- prints a bullet list of adapter names (for example `claude-code`, `cursor`, + `codex`, `windsurf`, `zed`). +- your host name appears in that list. if it does not, stop — pick the closest + supported adapter or wire `vouch serve` manually per + [transports.md](transports.md). + +## 2. wire the server + +replace `` with the name from step 1. + +```bash +vouch install-mcp +``` + +optional flags (all real — verified against `cli.py`): + +- `--tier T1|T2|T3|T4` — how much to install; tiers stack (default `T4`). +- `--path ` or `--target ` — project root to write into (default `.`). + +re-runs are flat-noop: expect lines containing `written`, `appended`, `merged`, +or `skipped`. you may run this command unconditionally on every session start. + +**assert:** exits 0; no `error:` lines. + +## 3. verify the kb.* surface + +```bash +vouch capabilities +``` + +**assert** on the JSON object: + +- `.name` is `"vouch"`. +- `.review_gated` is `true`. +- `.methods` includes at least: + - `kb.propose_claim` + - `kb.list_pending` + - `kb.approve` + +(`kb.approve` is exposed for trusted hosts; agents must still not call it — +see step 6.) + +## 4. create or locate a kb + +if `.vouch/` is missing: + +```bash +vouch init +``` + +**assert:** prints a path under the project and creates `.vouch/config.yaml`. + +if `.vouch/` already exists: + +```bash +vouch status +``` + +**assert:** prints `KB at …` with artifact counts and a `pending:` line. + +## 5. smoke test — propose (agent) + +create a throwaway citation file and register it: + +```bash +printf 'vouch agent install smoke test\n' > /tmp/vouch-agent-smoke.txt +vouch source add /tmp/vouch-agent-smoke.txt --title "agent smoke test" +``` + +**assert:** prints a 64-character hex source id (sha256 content address). + +propose a claim citing that source (replace ``): + +```bash +vouch propose-claim \ + --text "vouch agent install smoke test passed." \ + --source \ + --type observation \ + --confidence 0.9 +``` + +**assert:** prints a proposal id (for example `20260707-…`). + +confirm it is pending: + +```bash +vouch pending +``` + +**assert:** lists the proposal id from the previous step with `[claim]`. + +## 6. smoke test — approve (human only) + +**stop — this step is for the human reviewer, not the agent.** + +the agent must not run `vouch approve`, must not self-approve, and must not +hand-write files under `.vouch/claims/`, `.vouch/pages/`, or other `decided/` +paths. proposals live in `.vouch/proposed/` (gitignored) until a human decides. + +the human runs: + +```bash +vouch approve --reason "agent install smoke test" +vouch status +``` + +**assert:** `vouch status` shows the durable claim count incremented by one +compared to the count before step 5. + +## 7. what agents must never do + +- call `vouch approve` or `kb.approve` — the review gate is human-held. +- write yaml or markdown directly into `.vouch/claims/`, `.vouch/pages/`, or + other approved artifact directories. +- skip citation: every `propose-claim` needs at least one `--source` id. + +## where next + +- human-oriented walkthrough: [getting-started.md](getting-started.md) +- protocol and method shapes: [../SPEC.md](../SPEC.md) +- host-specific manifests: [../adapters/](../adapters/) diff --git a/docs/superpowers/plans/2026-07-09-artifact-delete.md b/docs/superpowers/plans/2026-07-09-artifact-delete.md new file mode 100644 index 00000000..39b4e836 --- /dev/null +++ b/docs/superpowers/plans/2026-07-09-artifact-delete.md @@ -0,0 +1,1066 @@ +# Review-Gated Artifact Delete Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add a hard-delete path for durable artifacts (claim, page, entity, relation) that routes entirely through the review gate — an agent files `kb.propose_delete`, a different reviewer approves via the existing `kb.approve`, and the artifact's file + index rows are removed. + +**Architecture:** Deletion is modeled as a new `ProposalKind.DELETE` proposal carrying `{target_kind, id, snapshot}`. `proposals.approve()` gains a branch that *removes* instead of *creates*. A shared `referenced_by()` helper enforces "block if referenced" at both propose and approve time. Storage gains pure-I/O `delete_*` unlinks; `index_db` gains a `deindex()` helper. One new method, `kb.propose_delete`, is mirrored across the four surfaces. + +**Tech Stack:** Python 3, pydantic v2, click (CLI), FastMCP (`@mcp.tool()`), SQLite FTS5 (`index_db`), pytest. + +## Global Constraints + +- Every write goes through `proposals.approve()`. No direct-mutation delete path. (CLAUDE.md north star.) +- `storage.py` is pure I/O — no business logic (no ref checks) in the `delete_*` methods. +- New `kb.*` method must be registered at all four sites: MCP tool (`server.py`), JSONL handler (`jsonl_server.py`), `METHODS` (`capabilities.py`), CLI (`cli.py`). `test_capabilities` enforces parity. +- CI gate (must stay green): `.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings`, `.venv/bin/python -m mypy src`, `.venv/bin/python -m ruff check src tests`. +- Conventional commits, lowercase summary ≤72 chars, lowercase body, **no `Co-Authored-By` trailer**. +- Stage specific files only — never `git add -A`. +- Commit messages via `git commit -F ` (the pre-commit hook rejects heredocs / some `-m` forms). Write the message to the scratchpad first. + +--- + +### Task 1: Storage delete methods (pure I/O) + +**Files:** +- Modify: `src/vouch/storage.py` (add four methods near `put_relation`, ~line 617) +- Test: `tests/test_delete.py` (create) + +**Interfaces:** +- Consumes: existing `self._claim_path`, `self._page_path`, `self._entity_path`, `self._relation_path`, and `ArtifactNotFoundError` (all already in `storage.py`). +- Produces: `KBStore.delete_claim(claim_id: str) -> None`, `.delete_page(page_id: str) -> None`, `.delete_entity(entity_id: str) -> None`, `.delete_relation(relation_id: str) -> None`. Each unlinks the file; raises `ArtifactNotFoundError` if absent. + +- [ ] **Step 1: Write the failing test** + +Create `tests/test_delete.py`: + +```python +"""Review-gated hard delete for durable artifacts (claim/page/entity/relation).""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from vouch.models import Claim, Entity, EntityType, Page, Relation, RelationType +from vouch.storage import ArtifactNotFoundError, KBStore + + +@pytest.fixture +def store(tmp_path: Path) -> KBStore: + return KBStore.init(tmp_path) + + +def _claim(store: KBStore, cid: str = "c1", text: str = "a claim") -> Claim: + src = store.put_source(b"src-bytes") + return store.put_claim(Claim(id=cid, text=text, evidence=[src.id])) + + +def test_delete_claim_removes_file(store: KBStore) -> None: + _claim(store, "c1") + assert store._claim_path("c1").exists() + store.delete_claim("c1") + assert not store._claim_path("c1").exists() + with pytest.raises(ArtifactNotFoundError): + store.get_claim("c1") + + +def test_delete_claim_missing_raises(store: KBStore) -> None: + with pytest.raises(ArtifactNotFoundError): + store.delete_claim("nope") + + +def test_delete_page_removes_file(store: KBStore) -> None: + store.put_page(Page(id="p1", title="P", body="hi")) + assert store._page_path("p1").exists() + store.delete_page("p1") + assert not store._page_path("p1").exists() + + +def test_delete_entity_removes_file(store: KBStore) -> None: + store.put_entity(Entity(id="e1", name="E", type=EntityType.CONCEPT)) + store.delete_entity("e1") + assert not store._entity_path("e1").exists() + + +def test_delete_relation_removes_file(store: KBStore) -> None: + _claim(store, "c1") + _claim(store, "c2") + rel = store.put_relation(Relation( + id="c1--supports--c2", source="c1", + relation=RelationType.SUPPORTS, target="c2", + )) + store.delete_relation(rel.id) + assert not store._relation_path(rel.id).exists() +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -q` +Expected: FAIL — `AttributeError: 'KBStore' object has no attribute 'delete_claim'`. + +- [ ] **Step 3: Write minimal implementation** + +In `src/vouch/storage.py`, add after `put_relation_idempotent` (i.e. after the relation write methods, before the page/source read helpers): + +```python + def delete_claim(self, claim_id: str) -> None: + """Remove a claim file. Pure I/O; ref checks live in `proposals`.""" + path = self._claim_path(claim_id) + if not path.exists(): + raise ArtifactNotFoundError(f"claim {claim_id}") + path.unlink() + + def delete_page(self, page_id: str) -> None: + """Remove a page file. Pure I/O; ref checks live in `proposals`.""" + path = self._page_path(page_id) + if not path.exists(): + raise ArtifactNotFoundError(f"page {page_id}") + path.unlink() + + def delete_entity(self, entity_id: str) -> None: + """Remove an entity file. Pure I/O; ref checks live in `proposals`.""" + path = self._entity_path(entity_id) + if not path.exists(): + raise ArtifactNotFoundError(f"entity {entity_id}") + path.unlink() + + def delete_relation(self, relation_id: str) -> None: + """Remove a relation file. Pure I/O; ref checks live in `proposals`.""" + path = self._relation_path(relation_id) + if not path.exists(): + raise ArtifactNotFoundError(f"relation {relation_id}") + path.unlink() +``` + +- [ ] **Step 4: Run tests + typecheck + lint** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -q` +Expected: PASS (5 passed). + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. + +- [ ] **Step 5: Commit** + +Write the message to `scratchpad/msg1.txt`: + +``` +feat(delete): add pure-io storage delete_* for four kinds + +delete_claim/page/entity/relation unlink the artifact file and raise +ArtifactNotFoundError if absent. no ref checks here — those live in the +proposals review gate. +``` + +```bash +git add src/vouch/storage.py tests/test_delete.py +git commit -F scratchpad/msg1.txt +``` + +--- + +### Task 2: `index_db.deindex()` helper + +**Files:** +- Modify: `src/vouch/index_db.py` (add near `index_claim`, ~line 180) +- Test: `tests/test_delete.py` (append) + +**Interfaces:** +- Consumes: existing tables `claims_fts`, `pages_fts`, `entities_fts`, `embedding_index(kind, id, ...)`, `prov_edges(src_id, dst_id, ...)`; existing `open_db`, `index_claim`, `index_prov_edge`. +- Produces: `index_db.deindex(conn: sqlite3.Connection, *, kind: str, id: str) -> None` — removes the FTS row (claim/page/entity only), the embedding row (any kind), and any prov edge touching `id`. + +- [ ] **Step 1: Write the failing test** + +Append to `tests/test_delete.py`: + +```python +from vouch import index_db + + +def test_deindex_removes_fts_and_prov(store: KBStore) -> None: + _claim(store, "c1", "searchable claim text") + with index_db.open_db(store.kb_dir) as conn: + index_db.index_claim( + conn, id="c1", text="searchable claim text", + type="observation", status="working", tags=[], + ) + index_db.index_prov_edge(conn, src_id="c1", dst_id="src-x", kind="cites") + index_db.index_prov_edge(conn, src_id="other", dst_id="c1", kind="cites") + # sanity: the fts row is present + with index_db.open_db(store.kb_dir) as conn: + pre = conn.execute("SELECT count(*) FROM claims_fts WHERE id='c1'").fetchone()[0] + assert pre == 1 + + with index_db.open_db(store.kb_dir) as conn: + index_db.deindex(conn, kind="claim", id="c1") + + with index_db.open_db(store.kb_dir) as conn: + assert conn.execute("SELECT count(*) FROM claims_fts WHERE id='c1'").fetchone()[0] == 0 + prov = conn.execute( + "SELECT count(*) FROM prov_edges WHERE src_id='c1' OR dst_id='c1'" + ).fetchone()[0] + assert prov == 0 + + +def test_deindex_relation_only_touches_embedding_and_prov(store: KBStore) -> None: + # relations have no FTS table; deindex must not raise for them. + with index_db.open_db(store.kb_dir) as conn: + index_db.index_prov_edge(conn, src_id="r1", dst_id="c2", kind="edge") + index_db.deindex(conn, kind="relation", id="r1") + assert conn.execute( + "SELECT count(*) FROM prov_edges WHERE src_id='r1' OR dst_id='r1'" + ).fetchone()[0] == 0 +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -k deindex -q` +Expected: FAIL — `AttributeError: module 'vouch.index_db' has no attribute 'deindex'`. + +- [ ] **Step 3: Write minimal implementation** + +In `src/vouch/index_db.py`, add directly after `index_claim` (before the provenance section comment): + +```python +def deindex(conn: sqlite3.Connection, *, kind: str, id: str) -> None: + """Remove every derived index row for a deleted artifact. + + FTS row for claim/page/entity (relations have no FTS table); the + embedding row for any kind (every put_* calls _embed_and_store, so an + embedding may exist for a relation too); and any provenance edge that + touches the id. prov_edges is otherwise rebuildable via + `kb.provenance_rebuild` — this keeps state.db consistent without a + full rebuild. + """ + if kind == "claim": + conn.execute("DELETE FROM claims_fts WHERE id = ?", (id,)) + elif kind == "page": + conn.execute("DELETE FROM pages_fts WHERE id = ?", (id,)) + elif kind == "entity": + conn.execute("DELETE FROM entities_fts WHERE id = ?", (id,)) + conn.execute( + "DELETE FROM embedding_index WHERE kind = ? AND id = ?", (kind, id) + ) + conn.execute( + "DELETE FROM prov_edges WHERE src_id = ? OR dst_id = ?", (id, id) + ) +``` + +- [ ] **Step 4: Run tests + typecheck + lint** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -q` +Expected: PASS (7 passed). + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. + +- [ ] **Step 5: Commit** + +Write to `scratchpad/msg2.txt`: + +``` +feat(delete): add index_db.deindex for removed artifacts + +drops the fts row (claim/page/entity), the embedding row (any kind), and +prov edges touching the id, so state.db stays consistent when an artifact +is hard-deleted. +``` + +```bash +git add src/vouch/index_db.py tests/test_delete.py +git commit -F scratchpad/msg2.txt +``` + +--- + +### Task 3: `ProposalKind.DELETE` + `referenced_by()` matrix helper + +**Files:** +- Modify: `src/vouch/models.py` (`ProposalKind` enum, ~line 392) +- Modify: `src/vouch/proposals.py` (add constants + `referenced_by`, near top-level helpers) +- Test: `tests/test_delete.py` (append) + +**Interfaces:** +- Consumes: `store.list_pages()`, `store.list_relations()`, `store.list_claims()` and the `Claim.supersedes/superseded_by/contradicts`, `Page.claims/entities`, `Relation.source/target` fields (all existing). +- Produces: + - `ProposalKind.DELETE = "delete"`. + - `proposals._DELETE_KINDS: set[str]` = `{"claim","page","entity","relation"}`. + - `proposals._DELETE_GETTERS: dict[str, str]` mapping kind → `KBStore` getter method name. + - `proposals.referenced_by(store, target_kind: str, target_id: str) -> list[str]` — human-readable descriptions of inbound referrers; empty list ⇒ deletable. Raises `ProposalError` on unknown kind. + +- [ ] **Step 1: Write the failing test** + +Append to `tests/test_delete.py`: + +```python +from vouch.models import ProposalKind +from vouch.proposals import ProposalError, referenced_by + + +def test_proposalkind_has_delete() -> None: + assert ProposalKind.DELETE.value == "delete" + + +def test_claim_referenced_by_page(store: KBStore) -> None: + _claim(store, "c1") + store.put_page(Page(id="p1", title="P", body="", claims=["c1"])) + refs = referenced_by(store, "claim", "c1") + assert any("p1" in r for r in refs) + + +def test_claim_referenced_by_relation_and_supersede(store: KBStore) -> None: + _claim(store, "c1") + _claim(store, "c2") + store.put_relation(Relation( + id="c2--supports--c1", source="c2", + relation=RelationType.SUPPORTS, target="c1", + )) + refs = referenced_by(store, "claim", "c1") + assert any("relation" in r for r in refs) + + +def test_unreferenced_claim_is_deletable(store: KBStore) -> None: + _claim(store, "lonely") + assert referenced_by(store, "claim", "lonely") == [] + + +def test_entity_referenced_by_claim(store: KBStore) -> None: + store.put_entity(Entity(id="e1", name="E", type=EntityType.CONCEPT)) + src = store.put_source(b"s") + store.put_claim(Claim(id="c1", text="mentions e1", evidence=[src.id], entities=["e1"])) + refs = referenced_by(store, "entity", "e1") + assert any("c1" in r for r in refs) + + +def test_relation_never_blocked(store: KBStore) -> None: + _claim(store, "c1") + _claim(store, "c2") + store.put_relation(Relation( + id="c1--supports--c2", source="c1", + relation=RelationType.SUPPORTS, target="c2", + )) + # nothing points at an edge → always deletable + assert referenced_by(store, "relation", "c1--supports--c2") == [] + + +def test_referenced_by_unknown_kind_raises(store: KBStore) -> None: + with pytest.raises(ProposalError): + referenced_by(store, "source", "x") +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -k "referenced_by or proposalkind_has_delete or deletable or never_blocked" -q` +Expected: FAIL — `AttributeError: DELETE` / `ImportError: cannot import name 'referenced_by'`. + +- [ ] **Step 3a: Add the enum member** + +In `src/vouch/models.py`, extend `ProposalKind`: + +```python +class ProposalKind(StrEnum): + CLAIM = "claim" + PAGE = "page" + ENTITY = "entity" + RELATION = "relation" + DELETE = "delete" +``` + +- [ ] **Step 3b: Add constants + helper in proposals.py** + +In `src/vouch/proposals.py`, add near the other module-level maps (e.g. just below `_ARTIFACT_GETTERS`, ~line 682): + +```python +_DELETE_KINDS = {"claim", "page", "entity", "relation"} + +_DELETE_GETTERS = { + "claim": "get_claim", + "page": "get_page", + "entity": "get_entity", + "relation": "get_relation", +} + + +def referenced_by(store: KBStore, target_kind: str, target_id: str) -> list[str]: + """Inbound referrers to `target_id` — the "block if referenced" gate. + + Returns human-readable descriptions of artifacts that point AT the + target. Only inbound refs count; outbound refs (what the target itself + points at) are never returned, because deleting the holder simply drops + its own pointers. An empty list means the artifact is safe to delete. + """ + if target_kind not in _DELETE_KINDS: + raise ProposalError( + f"unknown target_kind {target_kind!r}; expected one of " + f"{sorted(_DELETE_KINDS)}" + ) + refs: list[str] = [] + if target_kind == "claim": + for page in store.list_pages(): + if target_id in page.claims: + refs.append(f"page {page.id!r}") + for rel in store.list_relations(): + if target_id in (rel.source, rel.target): + refs.append(f"relation {rel.id!r}") + for claim in store.list_claims(): + if claim.id == target_id: + continue + if ( + target_id in claim.supersedes + or claim.superseded_by == target_id + or target_id in claim.contradicts + ): + refs.append(f"claim {claim.id!r}") + elif target_kind == "page": + for rel in store.list_relations(): + if target_id in (rel.source, rel.target): + refs.append(f"relation {rel.id!r}") + elif target_kind == "entity": + for claim in store.list_claims(): + if target_id in claim.entities: + refs.append(f"claim {claim.id!r}") + for page in store.list_pages(): + if target_id in page.entities: + refs.append(f"page {page.id!r}") + for rel in store.list_relations(): + if target_id in (rel.source, rel.target): + refs.append(f"relation {rel.id!r}") + # target_kind == "relation": edges have no inbound refs → refs stays empty + return refs +``` + +- [ ] **Step 4: Run tests + typecheck + lint** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -q` +Expected: PASS (all green). + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. + +- [ ] **Step 5: Commit** + +Write to `scratchpad/msg3.txt`: + +``` +feat(delete): add ProposalKind.DELETE and referenced_by matrix + +referenced_by returns inbound referrers per kind (claim: pages, relations, +supersede/contradict; page: relations; entity: claims, pages, relations; +relation: none). shared by the propose and approve delete gates. +``` + +```bash +git add src/vouch/models.py src/vouch/proposals.py tests/test_delete.py +git commit -F scratchpad/msg3.txt +``` + +--- + +### Task 4: `propose_delete()` + +**Files:** +- Modify: `src/vouch/proposals.py` (add `propose_delete` near the other `propose_*`, ~line 317) +- Test: `tests/test_delete.py` (append) + +**Interfaces:** +- Consumes: `referenced_by`, `_DELETE_KINDS`, `_DELETE_GETTERS`, `_file_proposal`, `ProposalKind.DELETE`, `ArtifactNotFoundError`. +- Produces: `proposals.propose_delete(store, *, target_kind: str, target_id: str, proposed_by: str, rationale: str | None = None, session_id: str | None = None, dry_run: bool = False) -> Proposal`. Payload shape `{"target_kind","id","snapshot"}`. Raises `ProposalError` for unknown kind, missing target, or referenced target. + +- [ ] **Step 1: Write the failing test** + +Append to `tests/test_delete.py`: + +```python +from vouch.models import ProposalStatus +from vouch.proposals import propose_delete + + +def test_propose_delete_files_pending(store: KBStore) -> None: + _claim(store, "c1", "delete me") + pr = propose_delete(store, target_kind="claim", target_id="c1", proposed_by="agent") + assert pr.kind is ProposalKind.DELETE + assert pr.status is ProposalStatus.PENDING + assert pr.payload["target_kind"] == "claim" + assert pr.payload["id"] == "c1" + assert pr.payload["snapshot"]["text"] == "delete me" + # still pending in the queue + assert any(p.id == pr.id for p in store.list_proposals(ProposalStatus.PENDING)) + + +def test_propose_delete_unknown_target_raises(store: KBStore) -> None: + with pytest.raises(ProposalError, match="unknown claim id"): + propose_delete(store, target_kind="claim", target_id="ghost", proposed_by="a") + + +def test_propose_delete_bad_kind_raises(store: KBStore) -> None: + with pytest.raises(ProposalError, match="unknown target_kind"): + propose_delete(store, target_kind="source", target_id="x", proposed_by="a") + + +def test_propose_delete_referenced_claim_blocked(store: KBStore) -> None: + _claim(store, "c1") + store.put_page(Page(id="p1", title="P", body="", claims=["c1"])) + with pytest.raises(ProposalError, match="referenced by"): + propose_delete(store, target_kind="claim", target_id="c1", proposed_by="a") + + +def test_propose_delete_claim_block_hints_supersede(store: KBStore) -> None: + _claim(store, "c1") + store.put_page(Page(id="p1", title="P", body="", claims=["c1"])) + with pytest.raises(ProposalError, match="supersede"): + propose_delete(store, target_kind="claim", target_id="c1", proposed_by="a") + + +def test_propose_delete_dry_run_writes_nothing(store: KBStore) -> None: + _claim(store, "c1") + pr = propose_delete( + store, target_kind="claim", target_id="c1", + proposed_by="a", dry_run=True, + ) + assert store.list_proposals(ProposalStatus.PENDING) == [] + assert pr.id # id is still returned for preview +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -k propose_delete -q` +Expected: FAIL — `ImportError: cannot import name 'propose_delete'`. + +- [ ] **Step 3: Write minimal implementation** + +In `src/vouch/proposals.py`, add after `propose_relation` (before the `# --- decisions ---` divider, ~line 317): + +```python +def propose_delete( + store: KBStore, + *, + target_kind: str, + target_id: str, + proposed_by: str, + rationale: str | None = None, + session_id: str | None = None, + dry_run: bool = False, +) -> Proposal: + """File a review-gated request to hard-delete a durable artifact. + + Blocked (at propose time, re-checked at approve) if the target is still + referenced by another artifact — the maintainer must supersede or remove + the referrers first. The full artifact is snapshotted into the payload so + the decided proposal and audit event record exactly what was removed. + """ + if target_kind not in _DELETE_KINDS: + raise ProposalError( + f"unknown target_kind {target_kind!r}; expected one of " + f"{sorted(_DELETE_KINDS)}" + ) + getter = getattr(store, _DELETE_GETTERS[target_kind]) + try: + artifact = getter(target_id) + except ArtifactNotFoundError as e: + raise ProposalError(f"unknown {target_kind} id: {target_id}") from e + refs = referenced_by(store, target_kind, target_id) + if refs: + hint = " (supersede it instead?)" if target_kind == "claim" else "" + raise ProposalError( + f"cannot delete {target_kind} {target_id}: referenced by " + + ", ".join(refs) + + hint + ) + payload = { + "target_kind": target_kind, + "id": target_id, + "snapshot": artifact.model_dump(mode="json"), + } + return _file_proposal( + store, kind=ProposalKind.DELETE, payload=payload, + proposed_by=proposed_by, session_id=session_id, + rationale=rationale, dry_run=dry_run, + ) +``` + +Note: `referenced_by`, `_DELETE_KINDS`, and `_DELETE_GETTERS` are defined lower in the file (Task 3). Python resolves them at call time, so a forward reference from `propose_delete` is fine. + +- [ ] **Step 4: Run tests + typecheck + lint** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -q` +Expected: PASS. + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. + +- [ ] **Step 5: Commit** + +Write to `scratchpad/msg4.txt`: + +``` +feat(delete): add proposals.propose_delete + +files a PENDING delete proposal for a claim/page/entity/relation, snapshots +the artifact into the payload, and refuses up front if the target is still +referenced (claims get a supersede hint). +``` + +```bash +git add src/vouch/proposals.py tests/test_delete.py +git commit -F scratchpad/msg4.txt +``` + +--- + +### Task 5: `approve()` DELETE branch + batch precheck + +**Files:** +- Modify: `src/vouch/proposals.py` (`approve` restructure + `_approve_delete`/`_reconstruct_deleted` helpers + `_payload_block_reason` branch) +- Test: `tests/test_delete.py` (append) + +**Interfaces:** +- Consumes: `store.delete_claim/page/entity/relation` (Task 1), `index_db.deindex` (Task 2), `referenced_by` (Task 3), `audit.log_event`, `Claim/Page/Entity/Relation` models. +- Produces: `approve()` handles `ProposalKind.DELETE` — removes the artifact, deindexes, logs a per-kind `.delete` audit event with the snapshot, returns the (former) artifact model. Idempotent when the artifact is already gone. `check_approvable` returns a block reason for a referenced delete target. + +- [ ] **Step 1: Write the failing test** + +Append to `tests/test_delete.py`: + +```python +from vouch import audit +from vouch.proposals import approve, check_approvable + + +def _propose_and_approve_delete(store: KBStore, kind: str, tid: str) -> None: + pr = propose_delete(store, target_kind=kind, target_id=tid, proposed_by="agent") + approve(store, pr.id, approved_by="reviewer") + + +def test_approve_delete_removes_claim_and_indexes(store: KBStore) -> None: + _claim(store, "c1", "gone soon") + _propose_and_approve_delete(store, "claim", "c1") + assert not store._claim_path("c1").exists() + with index_db.open_db(store.kb_dir) as conn: + assert conn.execute("SELECT count(*) FROM claims_fts WHERE id='c1'").fetchone()[0] == 0 + events = [e.event for e in audit.read_events(store.kb_dir)] + assert "claim.delete" in events + + +def test_approve_delete_page(store: KBStore) -> None: + store.put_page(Page(id="p1", title="P", body="x")) + _propose_and_approve_delete(store, "page", "p1") + assert not store._page_path("p1").exists() + + +def test_approve_delete_entity(store: KBStore) -> None: + store.put_entity(Entity(id="e1", name="E", type=EntityType.CONCEPT)) + _propose_and_approve_delete(store, "entity", "e1") + assert not store._entity_path("e1").exists() + + +def test_approve_delete_relation(store: KBStore) -> None: + _claim(store, "c1") + _claim(store, "c2") + store.put_relation(Relation( + id="c1--supports--c2", source="c1", + relation=RelationType.SUPPORTS, target="c2", + )) + _propose_and_approve_delete(store, "relation", "c1--supports--c2") + assert not store._relation_path("c1--supports--c2").exists() + + +def test_approve_rechecks_reference_added_after_propose(store: KBStore) -> None: + _claim(store, "c1") + pr = propose_delete(store, target_kind="claim", target_id="c1", proposed_by="agent") + # a page starts referencing c1 AFTER the proposal was filed + store.put_page(Page(id="p1", title="P", body="", claims=["c1"])) + with pytest.raises(ProposalError, match="still referenced"): + approve(store, pr.id, approved_by="reviewer") + # target survives, proposal stays pending + assert store._claim_path("c1").exists() + assert any(p.id == pr.id for p in store.list_proposals(ProposalStatus.PENDING)) + + +def test_approve_delete_idempotent_when_already_gone(store: KBStore) -> None: + _claim(store, "c1") + pr = propose_delete(store, target_kind="claim", target_id="c1", proposed_by="agent") + store.delete_claim("c1") # simulate a crash-retry: file already removed + result = approve(store, pr.id, approved_by="reviewer") + assert result.id == "c1" + # proposal is finalized (moved out of pending) + assert not any(p.id == pr.id for p in store.list_proposals(ProposalStatus.PENDING)) + + +def test_delete_forbids_self_approval(store: KBStore) -> None: + _claim(store, "c1") + pr = propose_delete(store, target_kind="claim", target_id="c1", proposed_by="same") + with pytest.raises(ProposalError, match="forbidden_self_approval"): + approve(store, pr.id, approved_by="same") + + +def test_check_approvable_flags_referenced_delete(store: KBStore) -> None: + _claim(store, "c1") + pr = propose_delete(store, target_kind="claim", target_id="c1", proposed_by="agent") + store.put_page(Page(id="p1", title="P", body="", claims=["c1"])) + reason = check_approvable(store, pr.id, approved_by="reviewer") + assert reason is not None and "referenced by" in reason +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -k "approve_delete or rechecks or idempotent or self_approval or check_approvable_flags" -q` +Expected: FAIL — the DELETE proposal falls through `approve()`'s `else: # RELATION` branch and errors constructing a `Relation`, or `_ensure_no_existing_artifact` KeyErrors. + +- [ ] **Step 3a: Exempt DELETE from the overwrite guard** + +In `src/vouch/proposals.py` `approve()`, change the guard (currently ~line 460): + +```python + if proposal.kind not in (ProposalKind.PAGE, ProposalKind.DELETE): + _ensure_no_existing_artifact(store, proposal.kind, payload["id"]) +``` + +- [ ] **Step 3b: Add the DELETE branch to the dispatch** + +In `approve()`, insert a branch before the final `else: # RELATION`: + +```python + elif proposal.kind == ProposalKind.DELETE: + result = _approve_delete(store, proposal, approved_by=approved_by) + else: # RELATION + rel = Relation(**payload) + store.put_relation(rel) + result = rel +``` + +- [ ] **Step 3c: Add the helper functions** + +In `src/vouch/proposals.py`, add near `referenced_by` (after it): + +```python +def _reconstruct_deleted( + target_kind: str, snapshot: dict[str, Any] +) -> Claim | Page | Entity | Relation: + """Rebuild a typed model from a delete proposal's snapshot. + + Used only on the idempotent path (artifact already gone) so the approve + surfaces still receive a `{kind, id}` result. + """ + if target_kind == "claim": + return Claim(**snapshot) + if target_kind == "page": + return Page(**snapshot) + if target_kind == "entity": + return Entity(**snapshot) + return Relation(**snapshot) + + +def _approve_delete( + store: KBStore, proposal: Proposal, *, approved_by: str +) -> Claim | Page | Entity | Relation: + """Execute an approved DELETE proposal: remove the artifact + index rows. + + Re-checks references at approve time (they may have appeared since the + proposal was filed). Idempotent: if the artifact is already gone, finalize + the proposal without erroring. + """ + payload = proposal.payload + target_kind = str(payload["target_kind"]) + target_id = str(payload["id"]) + snapshot = dict(payload.get("snapshot") or {}) + getter = getattr(store, _DELETE_GETTERS[target_kind]) + try: + artifact = getter(target_id) + except ArtifactNotFoundError: + return _reconstruct_deleted(target_kind, snapshot) + refs = referenced_by(store, target_kind, target_id) + if refs: + raise ProposalError( + f"cannot delete {target_kind} {target_id}: still referenced by " + + ", ".join(refs) + ) + deleter = getattr(store, f"delete_{target_kind}") + deleter(target_id) + with index_db.open_db(store.kb_dir) as conn: + index_db.deindex(conn, kind=target_kind, id=target_id) + audit.log_event( + store.kb_dir, event=f"{target_kind}.delete", actor=approved_by, + object_ids=[target_id], data={"snapshot": snapshot}, + ) + return artifact +``` + +- [ ] **Step 3d: Add the `_payload_block_reason` DELETE branch** + +In `_payload_block_reason`, add before the final `return None` (after the `ENTITY` branch, ~line 415): + +```python + elif proposal.kind == ProposalKind.DELETE: + target_kind = str(payload.get("target_kind", "")) + target_id = str(payload.get("id", "")) + if target_kind not in _DELETE_KINDS: + return f"invalid delete target_kind: {target_kind!r}" + getter = getattr(store, _DELETE_GETTERS[target_kind]) + try: + getter(target_id) + except ArtifactNotFoundError: + return None # already gone → idempotent approve is fine + refs = referenced_by(store, target_kind, target_id) + if refs: + return ( + f"cannot delete {target_kind} {target_id}: referenced by " + + ", ".join(refs) + ) +``` + +- [ ] **Step 4: Run the whole delete suite + full CI gate** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -q` +Expected: PASS. + +Run: `.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings` +Expected: PASS (no regressions). + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. + +- [ ] **Step 5: Commit** + +Write to `scratchpad/msg5.txt`: + +``` +feat(delete): execute delete proposals through approve() + +approve() now removes the artifact + index rows for a DELETE proposal, +re-checks references at the gate, logs a per-kind .delete audit event +with the snapshot, and is idempotent if the file is already gone. batch +precheck (check_approvable) reports a referenced target as unapprovable. +``` + +```bash +git add src/vouch/proposals.py tests/test_delete.py +git commit -F scratchpad/msg5.txt +``` + +--- + +### Task 6: Surface wiring (MCP + JSONL + CLI + METHODS) and parity + +**Files:** +- Modify: `src/vouch/capabilities.py` (`METHODS`) +- Modify: `src/vouch/server.py` (import + `kb_propose_delete` tool) +- Modify: `src/vouch/jsonl_server.py` (import + `_h_propose_delete` + `HANDLERS`) +- Modify: `src/vouch/cli.py` (import + `propose-delete` command) +- Test: `tests/test_delete.py` (append surface tests) + +**Interfaces:** +- Consumes: `proposals.propose_delete` (Task 4). +- Produces: `kb.propose_delete` reachable on all four surfaces; `METHODS` includes `"kb.propose_delete"`. + +- [ ] **Step 1: Write the failing test** + +Append to `tests/test_delete.py`: + +```python +from vouch import capabilities +from vouch.jsonl_server import handle_request + + +def test_method_registered_in_capabilities() -> None: + assert "kb.propose_delete" in capabilities.METHODS + + +def test_jsonl_propose_delete_end_to_end(store: KBStore, monkeypatch) -> None: + monkeypatch.chdir(store.root) + _claim(store, "c1", "kill via jsonl") + resp = handle_request({ + "id": "r1", + "method": "kb.propose_delete", + "params": {"target_kind": "claim", "target_id": "c1"}, + }) + assert resp["ok"] is True, resp + result = resp["result"] + assert result["kind"] == "delete" + assert result["status"] == "pending" + assert result["proposal_id"] +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_delete.py -k "method_registered or jsonl_propose_delete" -q` +Expected: FAIL — `"kb.propose_delete" not in METHODS` and the JSONL response carries an "unknown method" error. + +- [ ] **Step 3a: Register in `METHODS`** + +In `src/vouch/capabilities.py`, add after `"kb.propose_relation",`: + +```python + "kb.propose_relation", + "kb.propose_delete", +``` + +- [ ] **Step 3b: MCP tool** + +In `src/vouch/server.py`, add `propose_delete` to the `from .proposals import (...)` block (keep it alphabetical-ish, after `propose_claim`): + +```python + propose_claim, + propose_delete, + propose_entity, +``` + +Then add a tool alongside the other propose tools (near `kb_propose_relation`): + +```python +@mcp.tool() +def kb_propose_delete( + target_kind: str, target_id: str, rationale: str | None = None +) -> dict[str, Any]: + """Propose hard-deleting a durable artifact (claim/page/entity/relation). + + Files a PENDING delete request that a *different* reviewer approves via + kb.approve. Refused if the target is still referenced by another artifact. + """ + pr = propose_delete( + _store(), target_kind=target_kind, target_id=target_id, + proposed_by=_agent(), rationale=rationale, + ) + return {"proposal_id": pr.id, "status": pr.status.value, "kind": pr.kind.value} +``` + +- [ ] **Step 3c: JSONL handler** + +In `src/vouch/jsonl_server.py`, add `propose_delete` to the `from .proposals import (...)` block (after `propose_claim`): + +```python + propose_claim, + propose_delete, + propose_entity, +``` + +Add the handler near `_h_propose_relation`: + +```python +def _h_propose_delete(p: dict) -> dict: + pr = propose_delete( + _store(), + target_kind=p["target_kind"], + target_id=p["target_id"], + rationale=p.get("rationale"), + session_id=p.get("session_id"), + dry_run=bool(p.get("dry_run", False)), + proposed_by=_agent(), + ) + return { + "proposal_id": pr.id, + "status": pr.status.value, + "kind": pr.kind.value, + "dry_run": bool(p.get("dry_run", False)), + } +``` + +Register it in `HANDLERS` after `"kb.propose_relation": _h_propose_relation,`: + +```python + "kb.propose_relation": _h_propose_relation, + "kb.propose_delete": _h_propose_delete, +``` + +- [ ] **Step 3d: CLI command** + +In `src/vouch/cli.py`, add `propose_delete` to the first `from .proposals import (...)` block (after `propose_claim`): + +```python + propose_claim, + propose_delete, + propose_entity, +``` + +Add the command near the `supersede`/`archive` lifecycle commands (~line 1884): + +```python +@cli.command(name="propose-delete") +@click.argument( + "target_kind", + type=click.Choice(["claim", "page", "entity", "relation"]), +) +@click.argument("target_id") +@click.option("--rationale", default=None, help="why this should be deleted") +def propose_delete_cmd( + target_kind: str, target_id: str, rationale: str | None +) -> None: + """File a review-gated hard-delete request for an artifact. + + A different reviewer approves it with `vouch approve `. Refused if the + target is still referenced (supersede the claim instead, usually). + """ + store = _load_store() + with _cli_errors(): + pr = propose_delete( + store, target_kind=target_kind, target_id=target_id, + proposed_by=_whoami(), rationale=rationale, + ) + click.echo(f"filed delete proposal {pr.id} for {target_kind} {target_id}") +``` + +- [ ] **Step 4: Run the surface tests + full CI gate** + +Run: `.venv/bin/python -m pytest tests/test_delete.py tests/test_capabilities.py -q` +Expected: PASS (capabilities parity holds). + +Run: `.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings` +Expected: PASS. + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. + +- [ ] **Step 5: Manual smoke via CLI (verification)** + +Run: +```bash +cd "$(mktemp -d)" && .venv/bin/vouch init . >/dev/null 2>&1 || true +``` +Then in a scratch KB: register a source, propose+approve a claim, then: +```bash +vouch propose-delete claim +vouch list-pending # shows the delete proposal +vouch approve --reason "junk" +vouch read-claim # expect: not found +``` +Expected: the claim file is gone and `vouch audit` shows a `claim.delete` event. (Skip if a scratch KB is inconvenient; the pytest suite already exercises this end-to-end.) + +- [ ] **Step 6: Commit** + +Write to `scratchpad/msg6.txt`: + +``` +feat(delete): expose kb.propose_delete across all surfaces + +register the method on the MCP tool surface, the JSONL handler map, the +METHODS list, and the CLI (`vouch propose-delete `). approval +stays on the existing kb.approve. capabilities parity test passes. +``` + +```bash +git add src/vouch/capabilities.py src/vouch/server.py src/vouch/jsonl_server.py src/vouch/cli.py tests/test_delete.py +git commit -F scratchpad/msg6.txt +``` + +--- + +## Self-Review + +**1. Spec coverage** — every spec section maps to a task: +- object model (`ProposalKind.DELETE`, payload+snapshot) → Task 3 (enum) + Task 4 (payload). +- reference matrix → Task 3 (`referenced_by`). +- propose flow → Task 4. +- approve flow (skip overwrite guard, re-check refs, idempotent, per-kind audit) → Task 5. +- batch precheck → Task 5 (`_payload_block_reason`). +- storage layer → Task 1. +- index layer (`deindex`) → Task 2. +- four surfaces + parity → Task 6. +- reject/list/expire "no new code" → verified by the full-suite run in Tasks 5–6; nothing to build. +- out of scope (source/evidence, cascade, undo, web-ui) → not implemented, by design. + +**2. Placeholder scan** — no TBD/TODO; every code step shows complete code; every test step shows real assertions. + +**3. Type consistency** — `referenced_by(store, target_kind, target_id) -> list[str]`, `propose_delete(..., target_kind, target_id, ...) -> Proposal`, `_approve_delete(...) -> Claim|Page|Entity|Relation`, `deindex(conn, *, kind, id)`, `delete_(id)` names are used identically across the tasks that define and consume them. Payload keys `target_kind` / `id` / `snapshot` are consistent between `propose_delete`, `_approve_delete`, and `_payload_block_reason`. Surface params `target_kind` / `target_id` are consistent across MCP/JSONL/CLI. + +**Note for the implementer:** the JSONL/MCP/CLI params are `target_id` (not `id`) at the surface, but the *payload* key is `id`. That mapping is intentional — `propose_delete`'s `target_id` argument becomes `payload["id"]`. Don't "fix" one to match the other. diff --git a/docs/superpowers/plans/2026-07-09-session-split-summaries.md b/docs/superpowers/plans/2026-07-09-session-split-summaries.md new file mode 100644 index 00000000..495213fc --- /dev/null +++ b/docs/superpowers/plans/2026-07-09-session-split-summaries.md @@ -0,0 +1,1209 @@ +# Session-Split Summaries Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** When a session's captured activity is large, summarize it with an LLM into several topical `type: session` page proposals instead of one mechanical rollup — host-neutrally, with the review gate intact. + +**Architecture:** A new host-blind module `session_split.py` owns the buffer→pages pipeline (size gate → mechanical rollup *or* LLM topical split → PENDING proposals). `capture.finalize` becomes a thin wrapper that resolves the Claude-Code transcript intent and delegates. Shared LLM subprocess/parse plumbing is extracted from `compile.py` into `llm_draft.py`. A new `kb.summarize_session` method lets hosts without shell hooks trigger it. + +**Tech Stack:** Python 3, pydantic models, click CLI, pytest, mypy, ruff. LLM is a deployment-configured shell command (`capture.split.llm_cmd`, defaulting to `compile.llm_cmd`). + +## Global Constraints + +- Conventional commits, lowercase body, **no `Co-Authored-By` trailer** (checked in review). +- The commit hook rejects `git -m` and heredocs — write the message to `/tmp/vouch-commit-msg.txt` via `printf` and use `git commit -F`. +- Stage files **by name**; never `git add -A` (leaks `.claude/`, `webapp/`, WIP `.vouch/`). +- CI gate is exactly: `pytest tests/ -q --ignore=tests/embeddings`, `mypy src`, `ruff check src tests`. `make check` runs all three. +- **The review gate is load-bearing:** every produced page is filed via `proposals.propose_page` as a PENDING proposal. `approve()` is NEVER called in this feature. +- Split pages are always `page_type="session"` (forced in code) — never `concept`/`workflow`/`decision`. Sessions are feedstock, not compiled wiki pages. +- Work on branch `test` (current). Do not sweep the pre-existing WIP working tree into commits. + +--- + +## File Structure + +- **Create** `src/vouch/llm_draft.py` — shared `run_llm` + `parse_drafts` + fence-strip + `LLMDraftError`. +- **Create** `src/vouch/session_split.py` — host-blind `summarize()` core, `SplitConfig`, prompt builder, draft filer, audit. +- **Create** `tests/test_llm_draft.py`, `tests/test_session_split.py`. +- **Modify** `src/vouch/compile.py` — `run_llm`/`parse_drafts` become thin wrappers over `llm_draft`; drop now-unused imports. +- **Modify** `src/vouch/capture.py` — `finalize()` delegates to `session_split.summarize`; gains a `mode` param. +- **Modify** `src/vouch/capabilities.py` — add `"kb.summarize_session"` to `METHODS`. +- **Modify** `src/vouch/server.py` — add `kb_summarize_session` MCP tool. +- **Modify** `src/vouch/jsonl_server.py` — add `_h_summarize_session` + `HANDLERS` entry. +- **Modify** `src/vouch/cli.py` — `--split/--no-split` on `capture finalize`; new `capture summarize` command. +- **Modify** `src/vouch/storage.py` — add `capture.split` defaults to `_starter_config`. + +--- + +## Task 1: Extract shared LLM plumbing into `llm_draft.py` + +**Files:** +- Create: `src/vouch/llm_draft.py` +- Create: `tests/test_llm_draft.py` +- Modify: `src/vouch/compile.py` (imports; `run_llm`, `parse_drafts` bodies; remove local `_FENCE_RE`) + +**Interfaces:** +- Produces: `llm_draft.run_llm(llm_cmd: str, prompt: str, *, timeout_seconds: float, label: str = "llm_cmd") -> str`; `llm_draft.parse_drafts(raw: str, *, noun: str = "page") -> list[dict[str, Any]]`; `llm_draft.LLMDraftError(Exception)`; `llm_draft._FENCE_RE`. +- Consumes (in compile): nothing new; `compile.CompileError` still wraps failures so compile's public error contract and messages are unchanged. + +- [ ] **Step 1: Write the failing test** + +Create `tests/test_llm_draft.py`: + +```python +"""Shared LLM drafting plumbing used by compile and session_split.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest + +from vouch import llm_draft +from vouch.llm_draft import LLMDraftError, parse_drafts, run_llm + + +def _stub(tmp_path: Path, payload: str) -> str: + out = tmp_path / "out.txt" + out.write_text(payload, encoding="utf-8") + return f"cat {out}" + + +def test_run_llm_returns_stdout(tmp_path: Path) -> None: + cmd = _stub(tmp_path, '[{"title": "x", "body": "y"}]') + assert run_llm(cmd, "prompt", timeout_seconds=10.0).strip().startswith("[") + + +def test_run_llm_nonzero_raises_with_label(tmp_path: Path) -> None: + with pytest.raises(LLMDraftError, match="capture.split.llm_cmd failed"): + run_llm("false", "p", timeout_seconds=10.0, label="capture.split.llm_cmd") + + +def test_run_llm_timeout_raises(tmp_path: Path) -> None: + with pytest.raises(LLMDraftError, match="timed out"): + run_llm("sleep 5", "p", timeout_seconds=0.2) + + +def test_parse_drafts_strips_fence() -> None: + raw = '```json\n[{"title": "a", "body": "b"}]\n```' + assert parse_drafts(raw) == [{"title": "a", "body": "b"}] + + +def test_parse_drafts_bad_json_raises() -> None: + with pytest.raises(LLMDraftError, match="not valid JSON"): + parse_drafts("not json") + + +def test_parse_drafts_non_list_raises() -> None: + with pytest.raises(LLMDraftError, match="must be a JSON array"): + parse_drafts('{"title": "a"}') + + +def test_parse_drafts_non_dict_element_raises() -> None: + with pytest.raises(LLMDraftError, match="array of page objects"): + parse_drafts('["just a string"]') +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_llm_draft.py -q` +Expected: FAIL — `ModuleNotFoundError: No module named 'vouch.llm_draft'`. + +- [ ] **Step 3: Create `src/vouch/llm_draft.py`** + +```python +"""Shared LLM drafting plumbing for the review-gated compilers. + +Both `compile.py` (approved claims -> topic pages) and `session_split.py` +(session observations -> topical session pages) hand a deployment-configured +LLM command a prompt on stdin and parse a JSON array of drafts back. This +module is the shared subprocess + parse layer; each caller keeps its own +domain validation (compile verifies claim citations; session_split forces the +session page type). +""" + +from __future__ import annotations + +import json +import re +import subprocess +import tempfile +from typing import Any + +_FENCE_RE = re.compile(r"^```[a-zA-Z]*\n|\n```$") + + +class LLMDraftError(Exception): + """The LLM command could not run, or returned unusable output.""" + + +def run_llm( + llm_cmd: str, + prompt: str, + *, + timeout_seconds: float, + label: str = "llm_cmd", +) -> str: + """Run `llm_cmd` with `prompt` on stdin, in a throwaway temp cwd. + + `label` names the command in error messages so callers keep their own + config-key wording (e.g. "compile.llm_cmd"). Runs in a temp dir so an LLM + CLI that discovers per-project hooks/MCP from its cwd does not fire this + project's own pipeline while summarizing it. UTF-8 is forced on both pipe + directions — the default follows the locale (Latin-1 on some hosts), which + would crash on the first em-dash; `errors="replace"` surfaces a stray + invalid byte as a visible replacement char in review, not an exception. + """ + with tempfile.TemporaryDirectory(prefix="vouch-llm-") as tmp: + try: + proc = subprocess.run( + llm_cmd, shell=True, cwd=tmp, + input=prompt, capture_output=True, text=True, + encoding="utf-8", errors="replace", + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as e: + raise LLMDraftError(f"{label} timed out after {timeout_seconds:.0f}s") from e + if proc.returncode != 0: + detail = (proc.stderr or proc.stdout or "").strip()[:400] + raise LLMDraftError(f"{label} failed ({proc.returncode}): {detail}") + return proc.stdout + + +def parse_drafts(raw: str, *, noun: str = "page") -> list[dict[str, Any]]: + """Parse LLM stdout into a list of draft dicts. + + Strips a single markdown code fence if present. `noun` tunes error wording + ("page" -> "JSON array of pages"). Raises LLMDraftError on any shape + failure so callers can surface it as a clean, caller-visible message. + """ + text = _FENCE_RE.sub("", raw.strip()).strip() + try: + data = json.loads(text) + except json.JSONDecodeError as e: + raise LLMDraftError(f"compiler output is not valid JSON: {e}") from e + if not isinstance(data, list): + raise LLMDraftError(f"compiler output must be a JSON array of {noun}s") + for item in data: + if not isinstance(item, dict): + raise LLMDraftError( + f"compiler output must be a JSON array of {noun} objects, " + f"got element of type {type(item).__name__}" + ) + return list(data) +``` + +- [ ] **Step 4: Run the new test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_llm_draft.py -q` +Expected: PASS (7 passed). + +- [ ] **Step 5: Rewrite compile.py's `run_llm`/`parse_drafts` as wrappers** + +In `src/vouch/compile.py`, replace the entire bodies of `run_llm` (currently ~lines 182-210) and `parse_drafts` (~lines 213-230) and delete the module-level `_FENCE_RE` (~line 53) with: + +```python +def run_llm(llm_cmd: str, prompt: str, *, timeout_seconds: float) -> str: + """Run the configured LLM command with the prompt on stdin. + + Thin wrapper over `llm_draft.run_llm`, translating its error into the + `CompileError` compile callers already handle, and keeping the + "compile.llm_cmd …" wording in messages. + """ + try: + return llm_draft.run_llm( + llm_cmd, prompt, timeout_seconds=timeout_seconds, + label="compile.llm_cmd", + ) + except llm_draft.LLMDraftError as e: + raise CompileError(str(e)) from e + + +def parse_drafts(raw: str) -> list[dict[str, Any]]: + try: + return llm_draft.parse_drafts(raw, noun="page") + except llm_draft.LLMDraftError as e: + raise CompileError(str(e)) from e +``` + +Then fix the imports at the top of `compile.py`: add `from . import llm_draft` next to the other `from . import …` lines, and **remove** the now-unused `import json`, `import subprocess`, `import tempfile` (keep `import re` — it is still used by `_WIKILINK_RE`/`_CLAIM_MARKER_RE`). + +- [ ] **Step 6: Verify compile tests + lint still pass (unchanged behavior)** + +Run: `.venv/bin/python -m pytest tests/test_compile.py tests/test_llm_draft.py -q && .venv/bin/python -m ruff check src/vouch/compile.py src/vouch/llm_draft.py && .venv/bin/python -m mypy src/vouch/compile.py src/vouch/llm_draft.py` +Expected: all PASS. `test_compile.py`'s error-message assertions still match because the wrappers reproduce the exact strings. + +- [ ] **Step 7: Commit** + +```bash +printf '%s\n' 'refactor(compile): extract shared llm drafting into llm_draft' > /tmp/vouch-commit-msg.txt +git add src/vouch/llm_draft.py src/vouch/compile.py tests/test_llm_draft.py +git commit -F /tmp/vouch-commit-msg.txt +``` + +--- + +## Task 2: `SplitConfig` loaded from `capture.split` + +**Files:** +- Create: `src/vouch/session_split.py` (config portion only) +- Create: `tests/test_session_split.py` (config tests only) + +**Interfaces:** +- Produces: `session_split.SplitConfig` dataclass with fields `enabled: bool=True`, `llm_cmd: str|None=None`, `threshold_observations: int=40`, `max_pages: int=6`, `timeout_seconds: float=180.0`, `max_input_chars: int=60000`; `session_split.load_split_config(store: KBStore) -> SplitConfig`; `session_split.SplitConfigError(Exception)`. + +- [ ] **Step 1: Write the failing test** + +Create `tests/test_session_split.py`: + +```python +"""Host-blind session summarization: size gate, mechanical rollup, LLM split.""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from vouch import session_split +from vouch.session_split import SplitConfig, load_split_config +from vouch.storage import KBStore + + +@pytest.fixture +def store(tmp_path: Path) -> KBStore: + return KBStore.init(tmp_path) + + +def test_split_config_defaults(store: KBStore) -> None: + cfg = load_split_config(store) + assert cfg == SplitConfig() + assert cfg.threshold_observations == 40 + assert cfg.max_pages == 6 + assert cfg.enabled is True + + +def test_split_config_reads_override(store: KBStore) -> None: + store.config_path.write_text( + "capture:\n split:\n threshold_observations: 5\n max_pages: 2\n" + " llm_cmd: \"cat /dev/null\"\n", + encoding="utf-8", + ) + cfg = load_split_config(store) + assert cfg.threshold_observations == 5 + assert cfg.max_pages == 2 + assert cfg.llm_cmd == "cat /dev/null" + + +def test_split_config_malformed_yaml_falls_back(store: KBStore) -> None: + store.config_path.write_text("capture:\n split:\n - not-a-mapping\n", encoding="utf-8") + assert load_split_config(store) == SplitConfig() + + +def test_split_config_typo_coerces_to_default(store: KBStore) -> None: + store.config_path.write_text( + "capture:\n split:\n max_pages: six\n", encoding="utf-8" + ) + assert load_split_config(store).max_pages == 6 +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py -q` +Expected: FAIL — `ModuleNotFoundError: No module named 'vouch.session_split'`. + +- [ ] **Step 3: Create `src/vouch/session_split.py` (config only for now)** + +```python +"""Summarize a session's observation buffer into review-gated pages. + +Host-blind: reads only the normalized observation buffer +(`.vouch/captures/.jsonl`) that every host adapter writes via +`capture.observe`, never a host transcript. Small sessions get one mechanical +rollup page (reusing `capture.build_summary_body`); large sessions get an LLM +topical split into several `type: session` pages. Every page is a PENDING +proposal — `approve()` is never called. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import Any + +import yaml + +from .storage import KBStore + +logger = logging.getLogger(__name__) + +SPLIT_ACTOR = "session-split" + +DEFAULT_THRESHOLD_OBSERVATIONS = 40 +DEFAULT_MAX_PAGES = 6 +DEFAULT_TIMEOUT_SECONDS = 180.0 +DEFAULT_MAX_INPUT_CHARS = 60000 + + +class SplitConfigError(Exception): + """The split cannot run (no resolvable llm_cmd).""" + + +@dataclass(frozen=True) +class SplitConfig: + enabled: bool = True + llm_cmd: str | None = None + threshold_observations: int = DEFAULT_THRESHOLD_OBSERVATIONS + max_pages: int = DEFAULT_MAX_PAGES + timeout_seconds: float = DEFAULT_TIMEOUT_SECONDS + max_input_chars: int = DEFAULT_MAX_INPUT_CHARS + + +def _coerce(value: Any, default: Any, cast: Any) -> Any: + try: + return cast(value) + except (TypeError, ValueError): + return default + + +def load_split_config(store: KBStore) -> SplitConfig: + """Read `capture.split` from config.yaml; fall back to defaults.""" + try: + loaded = yaml.safe_load(store.config_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError): + return SplitConfig() + if not isinstance(loaded, dict): + return SplitConfig() + cap = loaded.get("capture") + raw = cap.get("split") if isinstance(cap, dict) else None + if not isinstance(raw, dict): + return SplitConfig() + llm_cmd = raw.get("llm_cmd") + return SplitConfig( + enabled=bool(raw.get("enabled", True)), + llm_cmd=str(llm_cmd) if llm_cmd else None, + threshold_observations=_coerce( + raw.get("threshold_observations", DEFAULT_THRESHOLD_OBSERVATIONS), + DEFAULT_THRESHOLD_OBSERVATIONS, int), + max_pages=_coerce(raw.get("max_pages", DEFAULT_MAX_PAGES), DEFAULT_MAX_PAGES, int), + timeout_seconds=_coerce( + raw.get("timeout_seconds", DEFAULT_TIMEOUT_SECONDS), + DEFAULT_TIMEOUT_SECONDS, float), + max_input_chars=_coerce( + raw.get("max_input_chars", DEFAULT_MAX_INPUT_CHARS), + DEFAULT_MAX_INPUT_CHARS, int), + ) +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py -q` +Expected: PASS (4 passed). + +- [ ] **Step 5: Commit** + +```bash +printf '%s\n' 'feat(session-split): add SplitConfig loaded from capture.split' > /tmp/vouch-commit-msg.txt +git add src/vouch/session_split.py tests/test_session_split.py +git commit -F /tmp/vouch-commit-msg.txt +``` + +--- + +## Task 3: `summarize()` mechanical pipeline + `finalize` delegation (behavior-preserving) + +**Files:** +- Modify: `src/vouch/session_split.py` (add `summarize`, `_propose_mechanical`) +- Modify: `src/vouch/capture.py` (`finalize` delegates; add `mode` param) +- Modify: `tests/test_session_split.py` (add pipeline tests) + +**Interfaces:** +- Produces: `session_split.summarize(store, session_id, *, intent=None, cwd=None, project=None, generated_at=None, mode="auto", config=None) -> dict[str, Any]`. Return dict keys: `captured:int`, `summary_proposal_id:str|None`, `summary_proposal_ids:list[str]`, `mode:str`, and `skipped:str` on skip paths. +- Consumes: `capture.load_config`, `capture.buffer_path`, `capture._read_observations`, `capture._git_changes`, `capture.build_summary_body`, `capture.CAPTURE_ACTOR`, `capture.CAPTURE_PAGE_TYPE`, `capture.CaptureConfig`; `proposals.propose_page`. +- `capture.finalize` keeps its signature and gains `mode: str = "auto"`, delegating to `summarize` via a deferred import (breaks the capture↔session_split cycle). + +- [ ] **Step 1: Write the failing tests** + +Append to `tests/test_session_split.py`: + +```python +def _observe(store: KBStore, sid: str, n: int, tool: str = "Edit") -> None: + from vouch import capture + for i in range(n): + capture.observe(store, sid, tool=tool, summary=f"{tool} file{i}.py", now=float(i)) + + +def test_below_min_skips_and_deletes_buffer(store: KBStore) -> None: + from vouch import capture + capture.observe(store, "s1", tool="Edit", summary="one", now=1.0) + res = session_split.summarize(store, "s1") + assert res["skipped"] == "below-min" + assert res["summary_proposal_ids"] == [] + assert not capture.buffer_path(store, "s1").exists() + + +def test_disabled_returns_skip(store: KBStore) -> None: + from vouch import capture + _observe(store, "s1", 5) + cfg = capture.CaptureConfig(enabled=False) + res = session_split.summarize(store, "s1", config=cfg) + assert res["skipped"] == "disabled" + + +def test_mechanical_single_page_below_threshold(store: KBStore) -> None: + from vouch.models import ProposalStatus + _observe(store, "s1", 5) # >= min (3), < threshold (40) + res = session_split.summarize(store, "s1", mode="auto") + assert res["mode"] == "mechanical" + assert len(res["summary_proposal_ids"]) == 1 + assert res["summary_proposal_id"] == res["summary_proposal_ids"][0] + pending = store.list_proposals(ProposalStatus.PENDING) + assert len(pending) == 1 + assert pending[0].payload["type"] == "session" + + +def test_finalize_still_returns_summary_proposal_id(store: KBStore) -> None: + from vouch import capture + _observe(store, "s1", 5) + res = capture.finalize(store, "s1", cwd=None, generated_at="2026-07-09T00:00:00Z") + assert "summary_proposal_id" in res + assert res["summary_proposal_id"] is not None + assert res["mode"] == "mechanical" +``` + +- [ ] **Step 2: Run tests to verify they fail** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py -k "below_min or disabled or mechanical or finalize_still" -q` +Expected: FAIL — `AttributeError: module 'vouch.session_split' has no attribute 'summarize'`. + +- [ ] **Step 3: Add `summarize` + `_propose_mechanical` to `session_split.py`** + +Add these imports to the top of `session_split.py` (next to the existing ones): + +```python +from pathlib import Path + +from . import capture +from .proposals import propose_page +``` + +Append to `session_split.py`: + +```python +def summarize( + store: KBStore, + session_id: str, + *, + intent: str | None = None, + cwd: Path | None = None, + project: str | None = None, + generated_at: str | None = None, + mode: str = "auto", + config: capture.CaptureConfig | None = None, +) -> dict[str, Any]: + """Roll a session buffer into PENDING page proposals. Never approves. + + `mode`: "auto" (size gate decides), "split" (force LLM), or "mechanical" + (force the single rollup). The buffer is deleted only after a page is + filed (or an explicit below-min skip), so a crash mid-run leaves it intact + for the next `finalize-all` sweep to retry. + """ + cfg = config or capture.load_config(store) + path = capture.buffer_path(store, session_id) + observations = capture._read_observations(path) + if not cfg.enabled: + return {"captured": len(observations), "summary_proposal_id": None, + "summary_proposal_ids": [], "mode": "skipped", "skipped": "disabled"} + if cwd is not None: + changed_files, git_stat = capture._git_changes(cwd) + else: + changed_files, git_stat = [], "" + total = len(observations) + len(changed_files) + if total < cfg.min_observations: + if path.exists(): + path.unlink() + return {"captured": total, "summary_proposal_id": None, + "summary_proposal_ids": [], "mode": "skipped", "skipped": "below-min"} + + # Task 4 inserts the LLM split branch here. + + pid = _propose_mechanical( + store, session_id, observations, changed_files, git_stat, + project=project, generated_at=generated_at, intent=intent, + ) + if path.exists(): + path.unlink() + return {"captured": total, "summary_proposal_id": pid, + "summary_proposal_ids": [pid], "mode": "mechanical"} + + +def _propose_mechanical( + store: KBStore, + session_id: str, + observations: list[dict[str, Any]], + changed_files: list[str], + git_stat: str, + *, + project: str | None, + generated_at: str | None, + intent: str | None, +) -> str: + """File the single mechanical rollup page, exactly as capture did before.""" + title, body = capture.build_summary_body( + session_id, observations, changed_files, git_stat, + project=project, generated_at=generated_at, first_prompt=intent, + ) + proposal = propose_page( + store, title=title, body=body, + page_type=capture.CAPTURE_PAGE_TYPE, + proposed_by=capture.CAPTURE_ACTOR, + session_id=session_id, + rationale="auto-captured session summary", + ) + return proposal.id +``` + +- [ ] **Step 4: Make `capture.finalize` delegate** + +In `src/vouch/capture.py`, replace the body of `finalize` (currently ~lines 315-369) with: + +```python +def finalize( + store: KBStore, + session_id: str, + *, + cwd: Path | None = None, + project: str | None = None, + generated_at: str | None = None, + transcript_path: Path | None = None, + mode: str = "auto", + config: CaptureConfig | None = None, +) -> dict[str, Any]: + """Roll a session buffer into PENDING summary proposal(s). No approve(). + + Claude-Code-facing wrapper: resolves the transcript's first user prompt + (its one host-specific enricher) and delegates to the host-blind + `session_split.summarize`. `mode` forwards "auto" | "split" | "mechanical". + """ + from . import session_split # deferred: breaks the capture<->session_split cycle + intent = ( + first_user_prompt(transcript_path) if transcript_path is not None else None + ) + return session_split.summarize( + store, session_id, intent=intent, cwd=cwd, project=project, + generated_at=generated_at, mode=mode, config=config, + ) +``` + +Leave `observe`, `summarize_tool`, `build_summary_body`, `first_user_prompt`, `finalize_all_except`, and the buffer helpers in place — `session_split` reuses them. + +- [ ] **Step 5: Run tests + lint + types** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py tests/test_capture.py -q && .venv/bin/python -m mypy src/vouch/session_split.py src/vouch/capture.py && .venv/bin/python -m ruff check src/vouch/session_split.py src/vouch/capture.py` +Expected: all PASS. `test_capture.py` stays green because the mechanical path reproduces the old actor/type/rationale and return keys. + +- [ ] **Step 6: Commit** + +```bash +printf '%s\n' 'refactor(capture): route finalize through session_split.summarize' > /tmp/vouch-commit-msg.txt +git add src/vouch/session_split.py src/vouch/capture.py tests/test_session_split.py +git commit -F /tmp/vouch-commit-msg.txt +``` + +--- + +## Task 4: LLM topical split + fallback + +**Files:** +- Modify: `src/vouch/session_split.py` (add split branch, `_propose_split`, `build_split_prompt`, `_file_drafts`, `_audit_split`) +- Modify: `tests/test_session_split.py` (add split tests) + +**Interfaces:** +- Consumes: `llm_draft.run_llm`, `llm_draft.parse_drafts`, `llm_draft.LLMDraftError`; `compile._pending_page_names`, `compile.load_config` (for the `compile.llm_cmd` fallback); `proposals._slugify`; `audit.log_event`; `store.list_pages`. +- Produces: internal `_propose_split(...) -> tuple[list[str], list[dict[str, Any]], bool]`; `summarize` now returns `mode="split"` with `dropped`/`truncated` on success, or `mode="fallback"` when the LLM path is attempted but yields nothing. + +- [ ] **Step 1: Write the failing tests** + +Append to `tests/test_session_split.py`: + +```python +def _stub_llm(tmp_path: Path, drafts: list[dict]) -> str: + out = tmp_path / "drafts.json" + out.write_text(json.dumps(drafts), encoding="utf-8") + return f"cat {out}" + + +def _config_with_split(store: KBStore, llm_cmd: str, threshold: int = 3, max_pages: int = 6) -> None: + store.config_path.write_text( + "capture:\n split:\n" + f" threshold_observations: {threshold}\n" + f" max_pages: {max_pages}\n" + f" llm_cmd: \"{llm_cmd}\"\n", + encoding="utf-8", + ) + + +def test_split_files_multiple_pending_session_pages(store: KBStore, tmp_path: Path) -> None: + from vouch.models import ProposalStatus + _observe(store, "s1", 5) + cmd = _stub_llm(tmp_path, [ + {"title": "refactored the audit writer", "body": "one thread of work " * 10}, + {"title": "fixed the ci locale bug", "body": "another thread of work " * 10}, + ]) + _config_with_split(store, cmd, threshold=3) + res = session_split.summarize(store, "s1", mode="auto") + assert res["mode"] == "split" + assert len(res["summary_proposal_ids"]) == 2 + pending = store.list_proposals(ProposalStatus.PENDING) + assert len(pending) == 2 + assert all(p.payload["type"] == "session" for p in pending) + assert all(p.proposed_by == session_split.SPLIT_ACTOR for p in pending) + assert store.list_pages() == [] # nothing durable — only proposed + + +def test_split_forces_session_type_even_if_llm_says_concept(store: KBStore, tmp_path: Path) -> None: + _observe(store, "s1", 5) + cmd = _stub_llm(tmp_path, [ + {"title": "a topic", "type": "concept", "body": "body " * 20}, + ]) + _config_with_split(store, cmd, threshold=3) + session_split.summarize(store, "s1", mode="auto") + from vouch.models import ProposalStatus + assert store.list_proposals(ProposalStatus.PENDING)[0].payload["type"] == "session" + + +def test_no_llm_cmd_falls_back_to_mechanical(store: KBStore) -> None: + _observe(store, "s1", 50) # over default threshold 40 + res = session_split.summarize(store, "s1", mode="auto") + assert res["mode"] == "fallback" + assert len(res["summary_proposal_ids"]) == 1 + + +def test_junk_llm_output_falls_back(store: KBStore) -> None: + _observe(store, "s1", 5) + _config_with_split(store, "echo not-json", threshold=3) + res = session_split.summarize(store, "s1", mode="auto") + assert res["mode"] == "fallback" + assert len(res["summary_proposal_ids"]) == 1 + + +def test_dedupe_drops_colliding_title(store: KBStore, tmp_path: Path) -> None: + from vouch.proposals import approve, propose_page + pr = propose_page(store, title="Existing Topic", body="b", page_type="concept", proposed_by="a") + approve(store, pr.id, approved_by="human-B") + _observe(store, "s1", 5) + cmd = _stub_llm(tmp_path, [ + {"title": "Existing Topic", "body": "dup " * 20}, + {"title": "Fresh Topic", "body": "fresh " * 20}, + ]) + _config_with_split(store, cmd, threshold=3) + res = session_split.summarize(store, "s1", mode="auto") + assert len(res["summary_proposal_ids"]) == 1 + assert any(d["reason"].startswith("title already") for d in res["dropped"]) + + +def test_cap_enforced(store: KBStore, tmp_path: Path) -> None: + _observe(store, "s1", 5) + drafts = [{"title": f"topic {i}", "body": "x " * 20} for i in range(5)] + cmd = _stub_llm(tmp_path, drafts) + _config_with_split(store, cmd, threshold=3, max_pages=2) + res = session_split.summarize(store, "s1", mode="auto") + assert len(res["summary_proposal_ids"]) == 2 + assert len([d for d in res["dropped"] if "over max_pages" in d["reason"]]) == 3 + + +def test_host_neutral_tool_names_do_not_crash(store: KBStore, tmp_path: Path) -> None: + from vouch import capture + for i, tool in enumerate(["fs.write", "shell.exec", "browser.open"]): + capture.observe(store, "s1", tool=tool, summary=f"{tool} did thing {i}", now=float(i)) + capture.observe(store, "s1", tool="fs.write", summary="one more", now=9.0) + cmd = _stub_llm(tmp_path, [{"title": "the work", "body": "did things " * 15}]) + _config_with_split(store, cmd, threshold=3) + res = session_split.summarize(store, "s1", mode="auto") + assert res["mode"] == "split" + + +def test_truncation_flagged_when_over_budget(store: KBStore, tmp_path: Path) -> None: + from vouch import capture + for i in range(50): + capture.observe(store, "s1", tool="Edit", summary="x" * 200, now=float(i)) + cmd = _stub_llm(tmp_path, [{"title": "t", "body": "b " * 20}]) + store.config_path.write_text( + "capture:\n split:\n threshold_observations: 3\n" + " max_input_chars: 500\n" + f" llm_cmd: \"{cmd}\"\n", + encoding="utf-8", + ) + res = session_split.summarize(store, "s1", mode="auto") + assert res["truncated"] is True +``` + +- [ ] **Step 2: Run tests to verify they fail** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py -k "split_files or forces_session or no_llm or junk or dedupe or cap_enforced or host_neutral or truncation" -q` +Expected: FAIL — the split path does not exist yet, so `mode` is never `"split"`/`"fallback"` (currently these observation counts route to `mechanical`, and default-threshold cases assert `fallback`). + +- [ ] **Step 3: Add the split branch + helpers to `session_split.py`** + +Add imports to the top of `session_split.py`: + +```python +from . import audit as audit_mod +from . import compile as compile_mod +from . import llm_draft +from .llm_draft import LLMDraftError +from .proposals import _slugify +``` + +Replace the `# Task 4 inserts the LLM split branch here.` marker in `summarize` with: + +```python + split_cfg = load_split_config(store) + want_split = mode == "split" or ( + mode == "auto" and split_cfg.enabled and total >= split_cfg.threshold_observations + ) + if mode != "mechanical" and want_split: + try: + ids, dropped, truncated = _propose_split( + store, session_id, observations, changed_files, git_stat, + intent=intent, split_cfg=split_cfg, + ) + if ids: + if path.exists(): + path.unlink() + return {"captured": total, "summary_proposal_id": ids[0], + "summary_proposal_ids": ids, "mode": "split", + "dropped": dropped, "truncated": truncated} + logger.warning( + "session_split: no valid drafts for %s; falling back to mechanical", + session_id, + ) + except (LLMDraftError, SplitConfigError) as e: + logger.warning( + "session_split: llm split failed for %s (%s); falling back", session_id, e + ) +``` + +And change the final mechanical return so a fallback is labeled distinctly. Replace the tail of `summarize` (the `_propose_mechanical` call and its return) with: + +```python + pid = _propose_mechanical( + store, session_id, observations, changed_files, git_stat, + project=project, generated_at=generated_at, intent=intent, + ) + if path.exists(): + path.unlink() + final_mode = "fallback" if (mode != "mechanical" and want_split) else "mechanical" + return {"captured": total, "summary_proposal_id": pid, + "summary_proposal_ids": [pid], "mode": final_mode} +``` + +Append the split helpers to `session_split.py`: + +```python +def _propose_split( + store: KBStore, + session_id: str, + observations: list[dict[str, Any]], + changed_files: list[str], + git_stat: str, + *, + intent: str | None, + split_cfg: SplitConfig, +) -> tuple[list[str], list[dict[str, Any]], bool]: + cmd = split_cfg.llm_cmd or compile_mod.load_config(store).llm_cmd + if not cmd: + raise SplitConfigError( + "capture.split.llm_cmd is not configured (and compile.llm_cmd is unset)" + ) + prompt, truncated = build_split_prompt( + store, observations, changed_files, git_stat, + intent=intent, max_pages=split_cfg.max_pages, + max_input_chars=split_cfg.max_input_chars, + ) + raw = llm_draft.run_llm( + cmd, prompt, timeout_seconds=split_cfg.timeout_seconds, + label="capture.split.llm_cmd", + ) + drafts = llm_draft.parse_drafts(raw, noun="page") + ids, dropped = _file_drafts(store, session_id, drafts, split_cfg.max_pages) + _audit_split(store, session_id, ids, dropped, len(observations), truncated) + return ids, dropped, truncated + + +def _render_obs(obs: dict[str, Any]) -> str: + tool = str(obs.get("tool", "")).strip() + summary = str(obs.get("summary", "")).strip() + files = obs.get("files") or [] + line = f"[{tool}] {summary}" if tool else summary + if files: + line += f" (files: {', '.join(str(f) for f in files[:5])})" + return line + + +def build_split_prompt( + store: KBStore, + observations: list[dict[str, Any]], + changed_files: list[str], + git_stat: str, + *, + intent: str | None, + max_pages: int, + max_input_chars: int, +) -> tuple[str, bool]: + """Assemble the host-neutral topical-split prompt. Returns (prompt, truncated). + + `tool` labels are opaque — the model clusters on the `summary` prose, so any + host's tool vocabulary works. If the rendered activity exceeds + `max_input_chars`, keep the most-recent observations that fit and prepend an + explicit elision note (no silent cap). + """ + obs_lines = [_render_obs(o) for o in observations] + truncated = False + if sum(len(x) + 1 for x in obs_lines) > max_input_chars: + truncated = True + kept: list[str] = [] + size = 0 + for line in reversed(obs_lines): + if size + len(line) + 1 > max_input_chars: + break + kept.append(line) + size += len(line) + 1 + obs_lines = list(reversed(kept)) + elided = len(observations) - len(obs_lines) + obs_lines.insert(0, f"(... {elided} older observations elided ...)") + + lines: list[str] = [ + "You are the session historian for this project's knowledge base. You", + "summarize one work session into a small set of durable, human-readable", + "session records — one per distinct thread of work.", + "", + ] + if intent: + lines += ["SESSION INTENT:", f" {intent}", ""] + lines += ["SESSION ACTIVITY (one line per observation, oldest first):"] + lines += [f"- {line}" for line in obs_lines] + lines += [""] + if changed_files: + lines += ["FILES CHANGED:"] + lines += [f"- {f}" for f in changed_files[:50]] + lines += [""] + if git_stat: + lines += ["GIT STAT:", "```", git_stat, "```", ""] + + pages = store.list_pages() + pending = compile_mod._pending_page_names(store) + taken = [f"- {p.title}" for p in pages] + [f"- {n} [pending]" for n in sorted(pending)] + lines += ["TAKEN TOPICS (do NOT redraft any of these):"] + lines += taken or ["- (none)"] + lines += [ + "", + "RULES", + f"- Cluster the activity into at most {max_pages} coherent TOPICS —", + " distinct threads of work in this session. Draft one page per topic.", + "- Each page needs a specific title (\"fixed the audit-log write race\",", + " not \"bug fixes\") and an 80-200 word markdown body summarizing that", + " thread of work.", + "- These are session records, NOT wiki topic pages: do NOT add", + " [claim: id] markers, and do NOT invent facts beyond the activity shown.", + "- Skip any topic already listed under TAKEN TOPICS.", + "", + "OUTPUT: print ONLY a JSON array, no code fences, no commentary.", + "Each element: {\"title\": str, \"body\": str}", + ] + return "\n".join(lines), truncated + + +def _file_drafts( + store: KBStore, + session_id: str, + drafts: list[dict[str, Any]], + max_pages: int, +) -> tuple[list[str], list[dict[str, Any]]]: + existing = store.list_pages() + taken = {p.title.strip().lower() for p in existing} + taken |= {p.id.strip().lower() for p in existing} + taken |= compile_mod._pending_page_names(store) + ids: list[str] = [] + dropped: list[dict[str, Any]] = [] + for i, draft in enumerate(drafts): + title = str(draft.get("title") or "").strip() + body = str(draft.get("body") or "").strip() + if not title: + dropped.append({"title": f"draft {i}", "reason": "draft has no title"}) + continue + if not body: + dropped.append({"title": title, "reason": "draft has no body"}) + continue + if len(ids) >= max_pages: + dropped.append({"title": title, "reason": f"over max_pages={max_pages}"}) + continue + if title.lower() in taken or _slugify(title) in taken: + dropped.append({"title": title, "reason": "title already exists or is pending"}) + continue + proposal = propose_page( + store, title=title, body=body, + page_type=capture.CAPTURE_PAGE_TYPE, # "session" — forced, ignore any LLM type + proposed_by=SPLIT_ACTOR, + tags=["session", "split"], + session_id=session_id, + metadata={"session_id": session_id}, + rationale=f"llm topical split of session {session_id}", + ) + ids.append(proposal.id) + taken.add(title.lower()) + taken.add(_slugify(title)) + return ids, dropped + + +def _audit_split( + store: KBStore, + session_id: str, + ids: list[str], + dropped: list[dict[str, Any]], + n_observations: int, + truncated: bool, +) -> None: + audit_mod.log_event( + store.kb_dir, event="session.split", actor=SPLIT_ACTOR, + object_ids=ids, + data={"proposed": len(ids), "dropped": len(dropped), + "observations": n_observations, "truncated": truncated}, + ) +``` + +- [ ] **Step 4: Run the split tests** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py -q` +Expected: PASS (all config + pipeline + split tests). + +- [ ] **Step 5: Lint + types + full capture/compile regression** + +Run: `.venv/bin/python -m mypy src/vouch/session_split.py && .venv/bin/python -m ruff check src/vouch/session_split.py && .venv/bin/python -m pytest tests/test_capture.py tests/test_compile.py -q` +Expected: all PASS. + +- [ ] **Step 6: Commit** + +```bash +printf '%s\n' 'feat(session-split): llm topical split for large sessions' > /tmp/vouch-commit-msg.txt +git add src/vouch/session_split.py tests/test_session_split.py +git commit -F /tmp/vouch-commit-msg.txt +``` + +--- + +## Task 5: Expose `kb.summarize_session` + CLI + starter config + +**Files:** +- Modify: `src/vouch/capabilities.py` (add method to `METHODS`) +- Modify: `src/vouch/jsonl_server.py` (`_h_summarize_session` + `HANDLERS`) +- Modify: `src/vouch/server.py` (`kb_summarize_session` MCP tool) +- Modify: `src/vouch/cli.py` (`--split/--no-split` on finalize; new `capture summarize`) +- Modify: `src/vouch/storage.py` (`_starter_config` split block) +- Modify: `tests/test_session_split.py` (surface tests) + +**Interfaces:** +- Produces: method `kb.summarize_session(session_id: str, mode: str = "auto") -> dict`. Registered in all four sites so `tests/test_capabilities.py::test_capabilities_matches_jsonl_handlers` passes. + +- [ ] **Step 1: Write the failing tests** + +Append to `tests/test_session_split.py`: + +```python +def test_kb_summarize_session_in_capabilities_and_handlers() -> None: + from vouch import capabilities + from vouch.jsonl_server import HANDLERS + assert "kb.summarize_session" in capabilities.METHODS + assert "kb.summarize_session" in HANDLERS + + +def test_jsonl_handler_summarizes(store: KBStore, monkeypatch: pytest.MonkeyPatch) -> None: + import vouch.jsonl_server as js + _observe(store, "s1", 5) + monkeypatch.setattr(js, "_store", lambda: store) + res = js.HANDLERS["kb.summarize_session"]({"session_id": "s1"}) + assert res["mode"] == "mechanical" + assert res["summary_proposal_id"] is not None + + +def test_starter_config_has_split_defaults() -> None: + from vouch.storage import _starter_config + split = _starter_config()["capture"]["split"] + assert split["threshold_observations"] == 40 + assert split["enabled"] is True +``` + +- [ ] **Step 2: Run tests to verify they fail** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py -k "capabilities_and_handlers or jsonl_handler or starter_config_has_split" -q` +Expected: FAIL — method not registered, no split block in starter config. + +- [ ] **Step 3: Register the method in `capabilities.py`** + +In `src/vouch/capabilities.py`, add one line to the `METHODS` list, right after `"kb.crystallize",`: + +```python + "kb.summarize_session", +``` + +- [ ] **Step 4: Add the JSONL handler** + +In `src/vouch/jsonl_server.py`, add the handler next to `_h_compile` (both are LLM ingest ops): + +```python +def _h_summarize_session(p: dict) -> dict: + from . import session_split + return session_split.summarize( + _store(), p["session_id"], mode=p.get("mode", "auto"), + ) +``` + +And add to the `HANDLERS` dict, next to `"kb.compile": _h_compile,`: + +```python + "kb.summarize_session": _h_summarize_session, +``` + +- [ ] **Step 5: Add the MCP tool** + +In `src/vouch/server.py`, add next to `kb_compile`: + +```python +@mcp.tool() +def kb_summarize_session( + session_id: str, + mode: str = "auto", +) -> dict[str, Any]: + """Summarize a captured session into PENDING page proposals. + + Reads the host-neutral observation buffer for `session_id` and files either + one mechanical rollup page (small sessions) or several LLM-drafted topical + `session` pages (large sessions). `mode` is "auto" | "split" | "mechanical". + Long-running when it splits (the LLM call is synchronous). Never approves. + """ + from . import session_split + return session_split.summarize(_store(), session_id, mode=mode) +``` + +- [ ] **Step 6: Add CLI `--split/--no-split` + `capture summarize`** + +In `src/vouch/cli.py`, add `from . import session_split` near the `capture as capture_mod` import. Then modify `capture_finalize_cmd` to accept the tri-state flag and forward a mode: + +```python +@capture.command("finalize") +@click.option("--session-id", default=None, help="Session id (else read from stdin payload).") +@click.option("--split/--no-split", "force", default=None, + help="Force LLM topical split or a single mechanical page (default: size-gated).") +def capture_finalize_cmd(session_id: str | None, force: bool | None) -> None: + """Roll a session buffer into PENDING summary proposal(s) (SessionEnd hook payload on stdin).""" + payload: dict[str, Any] = {} + if not sys.stdin.isatty(): + raw = sys.stdin.read() + if raw.strip(): + try: + loaded = json.loads(raw) + if isinstance(loaded, dict): + payload = loaded + except json.JSONDecodeError: + payload = {} + sid = session_id or str(payload.get("session_id") or "") + if not sid: + return + store = _capture_store() + if store is None: + return + cwd = Path(str(payload.get("cwd") or ".")).resolve() + transcript_raw = payload.get("transcript_path") + transcript = Path(str(transcript_raw)) if transcript_raw else None + mode = "auto" if force is None else ("split" if force else "mechanical") + result = capture_mod.finalize( + store, sid, cwd=cwd, project=cwd.name, + generated_at=datetime.now(UTC).isoformat(), + transcript_path=transcript, mode=mode, + ) + _emit_json(result) +``` + +And add a new command right after it: + +```python +@capture.command("summarize") +@click.argument("session_id") +@click.option("--split/--no-split", "force", default=None, + help="Force split or mechanical (default: size-gated auto).") +def capture_summarize_cmd(session_id: str, force: bool | None) -> None: + """Summarize a captured session into PENDING page proposals (size-gated).""" + store = _capture_store() + if store is None: + _emit_json({"error": "no KB found"}) + return + mode = "auto" if force is None else ("split" if force else "mechanical") + result = session_split.summarize( + store, session_id, mode=mode, generated_at=datetime.now(UTC).isoformat(), + ) + _emit_json(result) +``` + +- [ ] **Step 7: Add the split block to `_starter_config`** + +In `src/vouch/storage.py`, replace the `"capture"` block in `_starter_config` with: + +```python + "capture": { + # auto-capture agent sessions into pending summaries. + "enabled": True, + "min_observations": 3, + "split": { + # llm topical split for large sessions; llm_cmd falls back to + # compile.llm_cmd when null. see session_split.py. + "enabled": True, + "llm_cmd": None, + "threshold_observations": 40, + "max_pages": 6, + "timeout_seconds": 180, + "max_input_chars": 60000, + }, + }, +``` + +- [ ] **Step 8: Run surface tests + full suite** + +Run: `.venv/bin/python -m pytest tests/test_session_split.py tests/test_capabilities.py -q` +Expected: PASS, including `test_capabilities_matches_jsonl_handlers`. + +- [ ] **Step 9: Full CI gate** + +Run: `.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings && .venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: all green. + +- [ ] **Step 10: Commit** + +```bash +printf '%s\n' 'feat(session-split): expose kb.summarize_session across surfaces' > /tmp/vouch-commit-msg.txt +git add src/vouch/capabilities.py src/vouch/jsonl_server.py src/vouch/server.py src/vouch/cli.py src/vouch/storage.py tests/test_session_split.py +git commit -F /tmp/vouch-commit-msg.txt +``` + +--- + +## Self-Review + +**Spec coverage:** +- host-neutral IR / reads only the buffer → Task 3 (`summarize` reads `capture._read_observations`), Task 4 (opaque `tool` labels) ✓ +- `session_split.py` host-blind core → Tasks 3, 4 ✓ +- `llm_draft.py` extracted from compile → Task 1 ✓ +- three-tier size gate + mechanical default + fallback → Tasks 3, 4 ✓ +- split prompt/contract, no `[claim: id]`, `type` forced to session → Task 4 (`build_split_prompt`, `_file_drafts`) ✓ +- per-draft validation (title/body/dedupe/cap) → Task 4 (`_file_drafts`) ✓ +- huge-input guard with honest `truncated` flag → Task 4 (`build_split_prompt`) ✓ +- error handling → mechanical fallback, never raises to a hook → Task 4 (`summarize` try/except) ✓ +- audit `session.split` event → Task 4 (`_audit_split`) ✓ +- `kb.summarize_session` four sites + parity → Task 5 ✓ +- CLI `--split/--no-split` + `capture summarize` → Task 5 ✓ +- config block → Task 5 (`_starter_config`) + Task 2 (`load_split_config`) ✓ +- intent priority (`Session.task` → header → host parser → filename): partially — `finalize` supplies the CC transcript parser as `intent`; `Session.task` wiring is left to the caller passing `intent`. NOTE: v1 resolves intent at the `finalize`/CLI layer; the buffer-`intent`-header source is not yet written by any adapter, so only the transcript-parser and filename-fallback tiers are exercised. This matches the spec's "pluggable" framing and is not a gap for v1. + +**Placeholder scan:** No TBD/TODO; every code step shows complete code. The `# Task 4 inserts…` marker in Task 3 is replaced with real code in Task 4 Step 3. ✓ + +**Type consistency:** `summarize(...)` signature identical across Tasks 3–5; return dict keys (`captured`, `summary_proposal_id`, `summary_proposal_ids`, `mode`, `skipped`, `dropped`, `truncated`) consistent; `SplitConfig` fields match `load_split_config` and `_propose_split` usage; `capture.CAPTURE_PAGE_TYPE`/`CAPTURE_ACTOR` referenced consistently; `llm_draft.run_llm`/`parse_drafts` signatures match both compile wrappers and `_propose_split`. ✓ diff --git a/docs/superpowers/plans/2026-07-10-session-transcript-viewer.md b/docs/superpowers/plans/2026-07-10-session-transcript-viewer.md new file mode 100644 index 00000000..08bde17e --- /dev/null +++ b/docs/superpowers/plans/2026-07-10-session-transcript-viewer.md @@ -0,0 +1,1745 @@ +# Session Transcript Viewer Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Add a read-only Session Transcript Viewer to the vouch console that opens a captured Claude Code / Codex session and renders its full transcript (thinking, assistant text, tool calls with inputs + outputs, diffs, subagents), faithfully reproducing agentsview's rendering vocabulary. + +**Architecture:** A new `kb.session_transcript` RPC locates the raw agent JSONL on disk on demand, parses it into a normalized block schema, and returns it (degrading to vouch's compact capture observations when the raw file is gone). A new React **Sessions** tab lists sessions (`kb.list_sessions`, fanned out over scoped projects) and renders the selected session's blocks. No sync engine, no database — parsing happens per-open. + +**Tech Stack:** Backend — Python 3.11+, stdlib `json`/`pathlib`, `KBStore`, pytest. Frontend — React 19, TypeScript, Tailwind v4, `@tanstack/react-query`, `react-markdown`, `lucide-react`, vitest + Testing Library, Playwright. + +## Global Constraints + +- Repo: `vouch` at `/home/a/Dev/plind-junior/vouch`. Backend under `src/vouch/`, tests under `tests/`. Frontend under `webapp/`. +- Follow `vouch/AGENTS.md`, NOT agentsview's CLAUDE.md. Conventional commits `(): ` (types: feat|fix|refactor|test|docs|chore|perf|ci|style|build|revert), ≤72-char summary. +- **No `Co-Authored-By: ` trailer** in commits. No secrets or absolute machine paths as PII in commit messages. +- Do NOT switch/create branches without explicit user permission. Current branch is `test`; there are pre-existing uncommitted changes that are NOT ours — `git add` only our own files, never `git add -A`. +- Backend gates before every commit that touches Python: `.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings`, `.venv/bin/python -m mypy src`, `.venv/bin/python -m ruff check src tests`. +- Frontend gates before every commit that touches `webapp/`: `npm run test` and `npm run build` (from `webapp/`). +- Every new `kb.*` method MUST be added to both `capabilities.METHODS` and `jsonl_server.HANDLERS` or `test_capabilities_matches_jsonl_handlers` fails. +- Tests assert observable behavior, not implementation strings (testing-without-tautologies). Backend uses plain pytest `assert`. + +## Normalized transcript schema (the contract both sides share) + +Returned by `kb.session_transcript`. Tool results are paired into their tool_use block server-side. + +```jsonc +// available === true +{ + "available": true, + "source": { "agent": "claude", "path": "" }, + "session": { + "id": "…", "agent": "claude", + "cwd": "…"|null, "git_branch": "…"|null, "title": "…"|null, + "started_at": "ISO"|null, "ended_at": "ISO"|null, "model": "…"|null, + "tokens": { "input": 0, "output": 0, "cache_read": 0, "cache_creation": 0 } + }, + "messages": [ + { "role": "user"|"assistant", "id": "…"|null, "model": "…"|null, + "timestamp": "ISO"|null, + "tokens": { "input": 0, "output": 0, "cache_read": 0, "cache_creation": 0 }|null, + "blocks": [ + { "type": "text", "text": "…" }, + { "type": "thinking", "text": "…" }, + { "type": "tool_use", "id": "…", "name": "Bash", "input": {…}, + "result": { "content": "…", "is_error": false, + "subagent_session_id": "…"|null }|null } + ] } + ], + "truncated": false +} +// available === false +{ "available": false, "reason": "…", + "observations": [ { "ts": 0.0, "tool": "Edit", "summary": "Edited x.go", + "files": [...]|undefined, "cmd": "…"|undefined } ] } +``` + +--- + +## File Structure + +Backend: +- Create `src/vouch/transcript.py` — locator + Claude parser + Codex parser + `load_transcript` orchestrator. One responsibility: turn a session id into the normalized schema (or a degraded result). +- Modify `src/vouch/jsonl_server.py` — add `_h_session_transcript` handler + register in `HANDLERS`. +- Modify `src/vouch/capabilities.py` — add `"kb.session_transcript"` to `METHODS`. +- Create `tests/test_session_transcript.py` — parser/locator/handler/degradation tests + fixtures inline. + +Frontend (`webapp/`): +- Create `src/lib/transcript.ts` — TS types mirroring the schema + `fetchTranscript(conn, sessionId, agent?)`. +- Create `src/components/transcript/DiffView.tsx`, `CodeBlock.tsx`, `ThinkingBlock.tsx`, `ToolBlock.tsx`, `MessageBlock.tsx` — the block renderers. +- Create `src/views/TranscriptView.tsx` — fetches + renders one session (incl. degraded + subagent lazy-load). +- Create `src/views/SessionsView.tsx` — master–detail list + selection. +- Modify `src/App.tsx` — add `/sessions` route. +- Modify `src/components/Shell.tsx` — add Sessions nav item. +- Create colocated `*.test.tsx` for each component/view. +- Create `webapp/e2e/sessions.spec.ts` — smoke. + +--- + +## PHASE 1 — Backend: Claude locator, parser, RPC + +### Task 1: Claude file locator + +**Files:** +- Create: `src/vouch/transcript.py` +- Test: `tests/test_session_transcript.py` + +**Interfaces:** +- Produces: `find_claude_file(session_id: str) -> Path | None`; `_VALID_ID = re.compile(r"^[0-9a-fA-F-]{8,64}$")`; env override `VOUCH_CLAUDE_PROJECTS_DIR`. + +- [ ] **Step 1: Write the failing test** + +```python +# tests/test_session_transcript.py +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from vouch import transcript + + +def _write_jsonl(path: Path, records: list[dict]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("\n".join(json.dumps(r) for r in records) + "\n", encoding="utf-8") + + +def test_find_claude_file_top_level(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + root = tmp_path / "projects" + sid = "ad5d5e5f-0097-494c-8316-b01aed5dabf2" + f = root / "-home-a-Dev-agentsview" / f"{sid}.jsonl" + _write_jsonl(f, [{"type": "user", "message": {"role": "user", "content": "hi"}}]) + monkeypatch.setenv("VOUCH_CLAUDE_PROJECTS_DIR", str(root)) + assert transcript.find_claude_file(sid) == f + + +def test_find_claude_file_subagent(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + root = tmp_path / "projects" + parent = "11111111-1111-1111-1111-111111111111" + child = "22222222-2222-2222-2222-222222222222" + f = root / "-proj" / parent / "subagents" / "jobs" / f"{child}.jsonl" + _write_jsonl(f, [{"type": "assistant", "message": {"role": "assistant", "content": []}}]) + monkeypatch.setenv("VOUCH_CLAUDE_PROJECTS_DIR", str(root)) + assert transcript.find_claude_file(child) == f + + +def test_find_claude_file_rejects_bad_id(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("VOUCH_CLAUDE_PROJECTS_DIR", str(tmp_path)) + assert transcript.find_claude_file("../etc/passwd") is None + assert transcript.find_claude_file("*") is None + assert transcript.find_claude_file("") is None +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -q` +Expected: FAIL with `ModuleNotFoundError: No module named 'vouch.transcript'` / `AttributeError`. + +- [ ] **Step 3: Write minimal implementation** + +```python +# src/vouch/transcript.py +"""Locate and parse raw agent session transcripts on demand. + +Given a captured session id, find the raw JSONL the agent wrote +(Claude Code under ~/.claude/projects, Codex rollouts under +$CODEX_HOME/sessions) and normalize it into a block schema the vouch +console renders. Read-only: never writes to the KB. When the raw file is +gone we degrade to vouch's compact capture observations instead. +""" + +from __future__ import annotations + +import json +import os +import re +from pathlib import Path +from typing import Any + +# Session ids are UUID-shaped; reject anything else so a hostile id can't +# widen a glob or traverse out of the projects tree. +_VALID_ID = re.compile(r"^[0-9a-fA-F-]{8,64}$") + + +def _claude_projects_root() -> Path: + env = os.environ.get("VOUCH_CLAUDE_PROJECTS_DIR") + return Path(env) if env else Path.home() / ".claude" / "projects" + + +def find_claude_file(session_id: str) -> Path | None: + """The raw Claude Code JSONL for ``session_id``, or None. + + Claude names each session file ``.jsonl`` under a per-cwd project + dir; subagent transcripts live under ``/subagents/**``. The + file stem is the id, so a literal name match (no id interpolation into + a glob) locates it. + """ + if not _VALID_ID.match(session_id): + return None + root = _claude_projects_root() + if not root.is_dir(): + return None + name = f"{session_id}.jsonl" + for project in root.iterdir(): + if not project.is_dir(): + continue + top = project / name + if top.is_file(): + return top + for candidate in root.glob(f"*/*/subagents/**/{name}"): + if candidate.is_file(): + return candidate + return None +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -q` +Expected: PASS (3 passed). + +- [ ] **Step 5: Commit** + +```bash +git add src/vouch/transcript.py tests/test_session_transcript.py +git commit -m "feat(transcript): locate raw Claude session files by id" +``` + +--- + +### Task 2: Claude transcript parser + +**Files:** +- Modify: `src/vouch/transcript.py` +- Test: `tests/test_session_transcript.py` + +**Interfaces:** +- Consumes: nothing new. +- Produces: `parse_claude_transcript(path: Path, *, max_messages: int = 2000) -> dict[str, Any]` returning `{"session": {...}, "messages": [...], "truncated": bool}` per the schema; helper `_norm_tokens(usage: dict) -> dict`. + +- [ ] **Step 1: Write the failing test** + +```python +# append to tests/test_session_transcript.py + +_CLAUDE_LINES = [ + {"type": "user", "cwd": "/repo", "gitBranch": "main", + "timestamp": "2026-07-10T04:44:19.043Z", + "message": {"role": "user", "content": [{"type": "text", "text": "fix the bug"}]}}, + {"type": "ai-title", "aiTitle": "Fix the bug"}, + {"type": "assistant", "timestamp": "2026-07-10T04:44:35.759Z", + "message": {"id": "msg_1", "model": "claude-opus-4-8", "role": "assistant", + "content": [{"type": "thinking", "thinking": "let me look"}], + "usage": {"input_tokens": 100, "output_tokens": 10, + "cache_read_input_tokens": 5, "cache_creation_input_tokens": 2}}}, + {"type": "assistant", "timestamp": "2026-07-10T04:44:36.771Z", + "message": {"id": "msg_1", "model": "claude-opus-4-8", "role": "assistant", + "content": [{"type": "text", "text": "I'll edit it."}], + "usage": {"input_tokens": 100, "output_tokens": 10}}}, + {"type": "assistant", "timestamp": "2026-07-10T04:44:36.772Z", + "message": {"id": "msg_1", "model": "claude-opus-4-8", "role": "assistant", + "content": [{"type": "tool_use", "id": "tu_1", "name": "Bash", + "input": {"command": "go test ./..."}}], + "usage": {"input_tokens": 100, "output_tokens": 10}}}, + {"type": "user", + "message": {"role": "user", "content": [ + {"type": "tool_result", "tool_use_id": "tu_1", "content": "ok\n", "is_error": False}]}}, +] + + +def test_parse_claude_pairs_result_and_merges_by_message_id(tmp_path: Path) -> None: + f = tmp_path / "s.jsonl" + _write_jsonl(f, _CLAUDE_LINES) + out = transcript.parse_claude_transcript(f) + + assert out["session"]["cwd"] == "/repo" + assert out["session"]["git_branch"] == "main" + assert out["session"]["title"] == "Fix the bug" + assert out["session"]["model"] == "claude-opus-4-8" + assert out["truncated"] is False + + roles = [m["role"] for m in out["messages"]] + assert roles == ["user", "assistant"] # tool_result user entry is consumed, not a message + + # the three msg_1 assistant lines merged into one message, in order + a = out["messages"][1] + assert [b["type"] for b in a["blocks"]] == ["thinking", "text", "tool_use"] + tu = a["blocks"][2] + assert tu["name"] == "Bash" and tu["input"] == {"command": "go test ./..."} + assert tu["result"] == {"content": "ok\n", "is_error": False, "subagent_session_id": None} + assert a["tokens"] == {"input": 100, "output": 10, "cache_read": 5, "cache_creation": 2} + + +def test_parse_claude_truncates_at_cap(tmp_path: Path) -> None: + lines = [{"type": "user", "message": {"role": "user", "content": [{"type": "text", "text": f"m{i}"}]}} + for i in range(5)] + f = tmp_path / "big.jsonl" + _write_jsonl(f, lines) + out = transcript.parse_claude_transcript(f, max_messages=3) + assert out["truncated"] is True + assert len(out["messages"]) == 3 + + +def test_parse_claude_tolerates_malformed_lines(tmp_path: Path) -> None: + f = tmp_path / "m.jsonl" + f.write_text('{"bad json\n{"type":"user","message":{"role":"user","content":"hey"}}\n', encoding="utf-8") + out = transcript.parse_claude_transcript(f) + assert len(out["messages"]) == 1 + assert out["messages"][0]["blocks"] == [{"type": "text", "text": "hey"}] +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k parse_claude -q` +Expected: FAIL with `AttributeError: module 'vouch.transcript' has no attribute 'parse_claude_transcript'`. + +- [ ] **Step 3: Write minimal implementation** + +```python +# add to src/vouch/transcript.py + +def _norm_tokens(usage: dict[str, Any]) -> dict[str, int]: + def i(key: str) -> int: + v = usage.get(key) + return int(v) if isinstance(v, (int, float)) else 0 + return { + "input": i("input_tokens"), + "output": i("output_tokens"), + "cache_read": i("cache_read_input_tokens"), + "cache_creation": i("cache_creation_input_tokens"), + } + + +def _result_text(content: Any) -> str: + """tool_result.content is a string, or a list of {type:text,text} parts.""" + if isinstance(content, str): + return content + if isinstance(content, list): + parts = [str(p.get("text", "")) for p in content + if isinstance(p, dict) and p.get("type") == "text"] + if parts: + return "\n".join(parts) + return json.dumps(content, ensure_ascii=False) + if content is None: + return "" + return json.dumps(content, ensure_ascii=False) + + +def parse_claude_transcript(path: Path, *, max_messages: int = 2000) -> dict[str, Any]: + """Parse a Claude Code JSONL into the normalized transcript schema. + + Single forward pass: assistant content blocks that share a + ``message.id`` merge into one logical message; a later ``tool_result`` + (in a user entry) is paired into the matching ``tool_use`` block by id + and its user entry is not emitted as a standalone message. + """ + messages: list[dict[str, Any]] = [] + tool_by_id: dict[str, dict[str, Any]] = {} + session: dict[str, Any] = { + "id": path.stem, "agent": "claude", "cwd": None, "git_branch": None, + "title": None, "started_at": None, "ended_at": None, "model": None, + "tokens": {"input": 0, "output": 0, "cache_read": 0, "cache_creation": 0}, + } + truncated = False + current: dict[str, Any] | None = None + + def flush() -> None: + nonlocal current + if current is not None and current["blocks"]: + messages.append(current) + current = None + + with path.open(encoding="utf-8") as fh: + for raw in fh: + raw = raw.strip() + if not raw: + continue + try: + obj = json.loads(raw) + except json.JSONDecodeError: + continue + if not isinstance(obj, dict): + continue + if session["cwd"] is None and isinstance(obj.get("cwd"), str): + session["cwd"] = obj["cwd"] + if session["git_branch"] is None and isinstance(obj.get("gitBranch"), str): + session["git_branch"] = obj["gitBranch"] + ts = obj.get("timestamp") + if isinstance(ts, str): + if session["started_at"] is None: + session["started_at"] = ts + session["ended_at"] = ts + t = obj.get("type") + if t == "ai-title" and isinstance(obj.get("aiTitle"), str): + session["title"] = obj["aiTitle"] + continue + if t not in ("user", "assistant"): + continue + if len(messages) >= max_messages: + truncated = True + break + msg = obj.get("message") + if not isinstance(msg, dict): + continue + content = msg.get("content") + + if t == "assistant": + mid = msg.get("id") if isinstance(msg.get("id"), str) else None + if current is None or current.get("id") != mid: + flush() + usage = msg.get("usage") if isinstance(msg.get("usage"), dict) else {} + model = msg.get("model") if isinstance(msg.get("model"), str) else None + if model and session["model"] is None: + session["model"] = model + current = {"role": "assistant", "id": mid, "model": model, + "timestamp": ts if isinstance(ts, str) else None, + "tokens": _norm_tokens(usage), "blocks": []} + tok = current["tokens"] + for k in session["tokens"]: + session["tokens"][k] += tok[k] + parts = content if isinstance(content, list) else [] + for part in parts: + if not isinstance(part, dict): + continue + ptype = part.get("type") + if ptype == "thinking": + text = str(part.get("thinking", "")).strip() + if text: + current["blocks"].append({"type": "thinking", "text": text}) + elif ptype == "text": + text = str(part.get("text", "")).strip() + if text: + current["blocks"].append({"type": "text", "text": text}) + elif ptype == "tool_use": + tid = part.get("id") + block = {"type": "tool_use", "id": tid, + "name": str(part.get("name", "")), + "input": part.get("input") if isinstance(part.get("input"), dict) else {}, + "result": None} + current["blocks"].append(block) + if isinstance(tid, str): + tool_by_id[tid] = block + continue + + # user entry + flush() + if isinstance(content, str): + text = content.strip() + if text: + messages.append({"role": "user", "id": None, "model": None, + "timestamp": ts if isinstance(ts, str) else None, + "tokens": None, "blocks": [{"type": "text", "text": text}]}) + continue + parts = content if isinstance(content, list) else [] + user_blocks: list[dict[str, Any]] = [] + agent_id = None + tur = obj.get("toolUseResult") + if isinstance(tur, dict) and isinstance(tur.get("agentId"), str): + agent_id = tur["agentId"] + for part in parts: + if not isinstance(part, dict): + continue + if part.get("type") == "tool_result": + tid = part.get("tool_use_id") + block = tool_by_id.get(tid) if isinstance(tid, str) else None + if block is not None: + block["result"] = { + "content": _result_text(part.get("content")), + "is_error": bool(part.get("is_error", False)), + "subagent_session_id": agent_id, + } + elif part.get("type") == "text": + text = str(part.get("text", "")).strip() + if text: + user_blocks.append({"type": "text", "text": text}) + if user_blocks: + messages.append({"role": "user", "id": None, "model": None, + "timestamp": ts if isinstance(ts, str) else None, + "tokens": None, "blocks": user_blocks}) + flush() + return {"session": session, "messages": messages, "truncated": truncated} +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k parse_claude -q` +Expected: PASS (3 passed). + +- [ ] **Step 5: Run mypy + ruff** + +Run: `.venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: no errors. (Fix any typing nits inline, e.g. annotate `parts: list[Any]`.) + +- [ ] **Step 6: Commit** + +```bash +git add src/vouch/transcript.py tests/test_session_transcript.py +git commit -m "feat(transcript): parse Claude JSONL into normalized blocks" +``` + +--- + +### Task 3: `load_transcript` orchestrator with degradation + size cap + +**Files:** +- Modify: `src/vouch/transcript.py` +- Test: `tests/test_session_transcript.py` + +**Interfaces:** +- Consumes: `find_claude_file`, `parse_claude_transcript`, `capture.buffer_path`, `capture._read_observations`. +- Produces: `load_transcript(store: KBStore, session_id: str, *, agent: str | None = None) -> dict[str, Any]` returning either the available schema (with `source`) or the degraded schema. +- Constants: `MAX_FILE_BYTES = 25 * 1024 * 1024`, `MAX_MESSAGES = 2000`. + +- [ ] **Step 1: Write the failing test** + +```python +# append to tests/test_session_transcript.py +from vouch import capture +from vouch.storage import KBStore + + +@pytest.fixture +def store(tmp_path: Path) -> KBStore: + return KBStore.init(tmp_path) + + +def test_load_transcript_available(tmp_path: Path, store: KBStore, monkeypatch: pytest.MonkeyPatch) -> None: + root = tmp_path / "projects" + sid = "ad5d5e5f-0097-494c-8316-b01aed5dabf2" + f = root / "-repo" / f"{sid}.jsonl" + _write_jsonl(f, _CLAUDE_LINES) + monkeypatch.setenv("VOUCH_CLAUDE_PROJECTS_DIR", str(root)) + out = transcript.load_transcript(store, sid) + assert out["available"] is True + assert out["source"] == {"agent": "claude", "path": str(f)} + assert out["session"]["title"] == "Fix the bug" + + +def test_load_transcript_degrades_to_observations(store: KBStore, monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("VOUCH_CLAUDE_PROJECTS_DIR", str(store.kb_dir / "none")) + sid = "99999999-9999-9999-9999-999999999999" + capture.observe(store, sid, tool="Edit", summary="Edited x.go") + out = transcript.load_transcript(store, sid) + assert out["available"] is False + assert out["observations"][0]["tool"] == "Edit" + assert "reason" in out + + +def test_load_transcript_degrades_when_oversized(tmp_path: Path, store: KBStore, monkeypatch: pytest.MonkeyPatch) -> None: + root = tmp_path / "projects" + sid = "88888888-8888-8888-8888-888888888888" + f = root / "-repo" / f"{sid}.jsonl" + f.parent.mkdir(parents=True, exist_ok=True) + f.write_text("x" * 32, encoding="utf-8") + monkeypatch.setenv("VOUCH_CLAUDE_PROJECTS_DIR", str(root)) + monkeypatch.setattr(transcript, "MAX_FILE_BYTES", 16) + out = transcript.load_transcript(store, sid) + assert out["available"] is False + assert "too large" in out["reason"] +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k load_transcript -q` +Expected: FAIL (`load_transcript` undefined). + +- [ ] **Step 3: Write minimal implementation** + +```python +# add to src/vouch/transcript.py (add import at top: from .capture import _read_observations, buffer_path) +# and: from .storage import KBStore (TYPE_CHECKING is fine, but a runtime import is used by callers) + +MAX_FILE_BYTES = 25 * 1024 * 1024 +MAX_MESSAGES = 2000 + + +def _degraded(store: "KBStore", session_id: str, reason: str) -> dict[str, Any]: + obs = _read_observations(buffer_path(store, session_id)) + return {"available": False, "reason": reason, "observations": obs} + + +def load_transcript( + store: "KBStore", session_id: str, *, agent: str | None = None +) -> dict[str, Any]: + """Locate + parse the raw transcript for ``session_id``. + + ``agent`` restricts the search ("claude" | "codex"); when None both are + tried. Returns the normalized schema on success, or a degraded result + (compact capture observations) when the raw file is missing/too large. + """ + path: Path | None = None + source_agent = "" + if agent in (None, "claude"): + path = find_claude_file(session_id) + if path is not None: + source_agent = "claude" + if path is None and agent in (None, "codex"): + from . import codex_rollout + path = codex_rollout.find_rollout_by_session_id(session_id) + if path is not None: + source_agent = "codex" + if path is None: + return _degraded(store, session_id, f"raw transcript not found for session {session_id}") + try: + if path.stat().st_size > MAX_FILE_BYTES: + return _degraded(store, session_id, f"transcript too large to render ({path.stat().st_size} bytes)") + except OSError as e: + return _degraded(store, session_id, f"cannot read transcript: {e}") + + if source_agent == "claude": + parsed = parse_claude_transcript(path, max_messages=MAX_MESSAGES) + else: + parsed = parse_codex_transcript(path, max_messages=MAX_MESSAGES) + return {"available": True, "source": {"agent": source_agent, "path": str(path)}, **parsed} +``` + +Note: `parse_codex_transcript` lands in Task 9. Until then, guard the codex branch so Phase 1 imports cleanly — add a temporary stub at the bottom of the module that Task 9 replaces: + +```python +def parse_codex_transcript(path: Path, *, max_messages: int = 2000) -> dict[str, Any]: + raise NotImplementedError("codex transcript parsing lands in Task 9") +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k load_transcript -q` +Expected: PASS (3 passed). + +- [ ] **Step 5: Run full backend gates** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -q && .venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: all green. (Import `KBStore` under `TYPE_CHECKING` and quote the annotation to avoid a runtime cycle; `_read_observations` is a private-but-stable helper already used across the package.) + +- [ ] **Step 6: Commit** + +```bash +git add src/vouch/transcript.py tests/test_session_transcript.py +git commit -m "feat(transcript): orchestrate load with observation fallback" +``` + +--- + +### Task 4: `kb.session_transcript` RPC handler + +**Files:** +- Modify: `src/vouch/jsonl_server.py` (add handler near `_h_list_sessions` ~line 410; register in `HANDLERS` dict ~line 787) +- Modify: `src/vouch/capabilities.py` (add to `METHODS` list ~line 70, after `"kb.list_sessions"`) +- Test: `tests/test_session_transcript.py` + +**Interfaces:** +- Consumes: `transcript.load_transcript`, `_store()`. +- Produces: RPC method `kb.session_transcript(session_id, agent?)`. + +- [ ] **Step 1: Write the failing test** + +```python +# append to tests/test_session_transcript.py +from vouch.capabilities import capabilities +from vouch.jsonl_server import HANDLERS, handle_request + + +def test_capabilities_advertises_session_transcript() -> None: + assert "kb.session_transcript" in capabilities().methods + assert "kb.session_transcript" in HANDLERS + + +def test_handler_missing_session_id_is_missing_param() -> None: + resp = handle_request({"id": "1", "method": "kb.session_transcript", "params": {}}) + assert resp["ok"] is False + assert resp["error"]["code"] == "missing_param" + + +def test_handler_bad_agent_is_invalid_request() -> None: + resp = handle_request({"id": "2", "method": "kb.session_transcript", + "params": {"session_id": "11111111-1111-1111-1111-111111111111", "agent": "grok"}}) + assert resp["ok"] is False + assert resp["error"]["code"] == "invalid_request" + + +def test_handler_returns_degraded_when_absent(monkeypatch: pytest.MonkeyPatch) -> None: + resp = handle_request({"id": "3", "method": "kb.session_transcript", + "params": {"session_id": "11111111-1111-1111-1111-111111111111"}}) + assert resp["ok"] is True + assert resp["result"]["available"] is False +``` + +Note: `test_handler_returns_degraded_when_absent` runs against the real KB discovered by `_store()`. It asserts only the shape (degraded), which holds regardless of on-disk files, because that id will not exist. + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k "handler or advertises" -q` +Expected: FAIL — `test_capabilities_advertises_session_transcript` fails and `test_capabilities_matches_jsonl_handlers` (existing) fails once you add to only one list; handler tests fail with `method_not_found`. + +- [ ] **Step 3: Write minimal implementation** + +In `src/vouch/capabilities.py`, add to the `METHODS` list right after `"kb.list_sessions",`: + +```python + "kb.session_transcript", +``` + +In `src/vouch/jsonl_server.py`, add the handler (place it just after `_h_list_sessions`): + +```python +def _h_session_transcript(p: dict) -> dict: + from . import transcript + session_id = p["session_id"] + agent = p.get("agent") + if agent is not None and agent not in ("claude", "codex"): + raise ValueError(f"unknown agent: {agent!r} (expected 'claude' or 'codex')") + return transcript.load_transcript(_store(), session_id, agent=agent) +``` + +Register it in the `HANDLERS` dict, right after the `"kb.list_sessions": _h_list_sessions,` line: + +```python + "kb.session_transcript": _h_session_transcript, +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -q` +Expected: PASS (all). Also run `.venv/bin/python -m pytest tests/test_capabilities.py -q` → PASS. + +- [ ] **Step 5: Full backend gates** + +Run: `.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings && .venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests` +Expected: all green. + +- [ ] **Step 6: Commit** + +```bash +git add src/vouch/jsonl_server.py src/vouch/capabilities.py tests/test_session_transcript.py +git commit -m "feat(transcript): expose kb.session_transcript RPC" +``` + +--- + +## PHASE 2 — Frontend: rendering + Sessions view + +Work from `webapp/`. Run `npm install` once if `node_modules` is absent. + +### Task 5: Transcript client types + fetch + +**Files:** +- Create: `webapp/src/lib/transcript.ts` +- Test: `webapp/src/lib/transcript.test.ts` + +**Interfaces:** +- Produces: types `TranscriptBlock`, `TranscriptMessage`, `SessionMeta`, `Transcript` (union of available/degraded); `fetchTranscript(conn, sessionId, agent?) -> Promise`. + +- [ ] **Step 1: Write the failing test** + +```ts +// webapp/src/lib/transcript.test.ts +import { describe, expect, it, vi } from 'vitest' + +vi.mock('./rpc', () => ({ rpc: vi.fn() })) +import { rpc } from './rpc' +import { fetchTranscript } from './transcript' +import type { VouchConnectionInfo } from './types' + +const conn: VouchConnectionInfo = { endpoint: 'http://127.0.0.1:8731' } + +describe('fetchTranscript', () => { + it('calls kb.session_transcript with session id + agent', async () => { + vi.mocked(rpc).mockResolvedValue({ available: false, reason: 'x', observations: [] }) + await fetchTranscript(conn, 'sid-1', 'claude') + expect(rpc).toHaveBeenCalledWith(conn, 'kb.session_transcript', { session_id: 'sid-1', agent: 'claude' }) + }) + + it('omits agent when not given', async () => { + vi.mocked(rpc).mockResolvedValue({ available: false, reason: 'x', observations: [] }) + await fetchTranscript(conn, 'sid-2') + expect(rpc).toHaveBeenCalledWith(conn, 'kb.session_transcript', { session_id: 'sid-2' }) + }) +}) +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `npm run test -- transcript.test.ts` +Expected: FAIL (module `./transcript` not found). + +- [ ] **Step 3: Write minimal implementation** + +```ts +// webapp/src/lib/transcript.ts +import { rpc } from './rpc' +import type { VouchConnectionInfo } from './types' + +export interface Tokens { input: number; output: number; cache_read: number; cache_creation: number } + +export interface ToolResult { + content: string + is_error: boolean + subagent_session_id: string | null +} + +export type TranscriptBlock = + | { type: 'text'; text: string } + | { type: 'thinking'; text: string } + | { type: 'tool_use'; id: string | null; name: string; input: Record; result: ToolResult | null } + +export interface TranscriptMessage { + role: 'user' | 'assistant' + id: string | null + model: string | null + timestamp: string | null + tokens: Tokens | null + blocks: TranscriptBlock[] +} + +export interface SessionMeta { + id: string + agent: string + cwd: string | null + git_branch: string | null + title: string | null + started_at: string | null + ended_at: string | null + model: string | null + tokens: Tokens +} + +export interface Observation { ts: number; tool: string; summary: string; files?: string[]; cmd?: string } + +export type Transcript = + | { available: true; source: { agent: string; path: string }; session: SessionMeta; messages: TranscriptMessage[]; truncated: boolean } + | { available: false; reason: string; observations: Observation[] } + +export function fetchTranscript( + conn: VouchConnectionInfo, + sessionId: string, + agent?: string, +): Promise { + const params: Record = { session_id: sessionId } + if (agent) params.agent = agent + return rpc(conn, 'kb.session_transcript', params) +} +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `npm run test -- transcript.test.ts` +Expected: PASS. + +- [ ] **Step 5: Commit** + +```bash +git add webapp/src/lib/transcript.ts webapp/src/lib/transcript.test.ts +git commit -m "feat(webapp): transcript client types and fetch" +``` + +--- + +### Task 6: Leaf block renderers (DiffView, CodeBlock, ThinkingBlock) + +**Files:** +- Create: `webapp/src/components/transcript/DiffView.tsx`, `CodeBlock.tsx`, `ThinkingBlock.tsx` +- Test: `webapp/src/components/transcript/blocks.test.tsx` + +**Interfaces:** +- Produces: ``; ``; ``. + +- [ ] **Step 1: Write the failing test** + +```tsx +// webapp/src/components/transcript/blocks.test.tsx +import { render, screen } from '@testing-library/react' +import userEvent from '@testing-library/user-event' +import { describe, expect, it } from 'vitest' +import { DiffView } from './DiffView' +import { ThinkingBlock } from './ThinkingBlock' + +describe('DiffView', () => { + it('tags added and removed lines', () => { + const { container } = render() + expect(container.querySelector('.diff-add')?.textContent).toContain('+new') + expect(container.querySelector('.diff-del')?.textContent).toContain('-old') + expect(container.querySelector('.diff-hunk')?.textContent).toContain('@@') + }) +}) + +describe('ThinkingBlock', () => { + it('is collapsed by default and expands on click', async () => { + render() + expect(screen.queryByText('secret reasoning')).toBeNull() + await userEvent.click(screen.getByRole('button', { name: /thinking/i })) + expect(screen.getByText('secret reasoning')).toBeInTheDocument() + }) +}) +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `npm run test -- blocks.test.tsx` +Expected: FAIL (modules not found). + +- [ ] **Step 3: Write minimal implementation** + +```tsx +// webapp/src/components/transcript/DiffView.tsx +export function DiffView({ text }: { text: string }) { + const lines = text.replace(/\n$/, '').split('\n') + return ( +
+ {lines.map((line, i) => { + const cls = line.startsWith('@@') + ? 'diff-hunk text-accent-2' + : line.startsWith('+') + ? 'diff-add bg-ok/10 text-ok' + : line.startsWith('-') + ? 'diff-del bg-accent/10 text-accent-2' + : 'diff-ctx text-ink-2' + return ( +
+ {line || ' '} +
+ ) + })} +
+ ) +} +``` + +```tsx +// webapp/src/components/transcript/CodeBlock.tsx +export function CodeBlock({ code, lang }: { code: string; lang?: string }) { + return ( +
+ {lang && ( +
+ {lang} +
+ )} +
{code}
+
+ ) +} +``` + +```tsx +// webapp/src/components/transcript/ThinkingBlock.tsx +import { Brain, ChevronRight } from 'lucide-react' +import { useState } from 'react' + +export function ThinkingBlock({ text }: { text: string }) { + const [open, setOpen] = useState(false) + return ( +
+ + {open &&
{text}
} +
+ ) +} +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `npm run test -- blocks.test.tsx` +Expected: PASS. + +- [ ] **Step 5: Commit** + +```bash +git add webapp/src/components/transcript/DiffView.tsx webapp/src/components/transcript/CodeBlock.tsx webapp/src/components/transcript/ThinkingBlock.tsx webapp/src/components/transcript/blocks.test.tsx +git commit -m "feat(webapp): thinking, diff, and code block renderers" +``` + +--- + +### Task 7: ToolBlock (per-tool rendering + collapsible + result) + +**Files:** +- Create: `webapp/src/components/transcript/ToolBlock.tsx` +- Test: `webapp/src/components/transcript/ToolBlock.test.tsx` + +**Interfaces:** +- Consumes: `DiffView`, `CodeBlock`, block type from `lib/transcript`, optional `onOpenSubagent(sessionId)` callback. +- Produces: ` void} />`. + +- [ ] **Step 1: Write the failing test** + +```tsx +// webapp/src/components/transcript/ToolBlock.test.tsx +import { render, screen } from '@testing-library/react' +import userEvent from '@testing-library/user-event' +import { describe, expect, it, vi } from 'vitest' +import { ToolBlock } from './ToolBlock' +import type { TranscriptBlock } from '../../lib/transcript' + +function tool(over: Partial>): Extract { + return { type: 'tool_use', id: 't1', name: 'Bash', input: {}, result: null, ...over } +} + +describe('ToolBlock', () => { + it('shows the tool name and a Bash command header', () => { + render() + expect(screen.getByText('Bash')).toBeInTheDocument() + expect(screen.getByText('go test ./...')).toBeInTheDocument() + }) + + it('renders a diff for Edit results and reveals output on expand', async () => { + const block = tool({ + name: 'Edit', + input: { file_path: '/x.go' }, + result: { content: '@@ -1 +1 @@\n-a\n+b', is_error: false, subagent_session_id: null }, + }) + const { container } = render() + await userEvent.click(screen.getByRole('button', { name: /Edit/i })) + expect(container.querySelector('.diff-add')).not.toBeNull() + }) + + it('marks errored results', async () => { + const block = tool({ result: { content: 'boom', is_error: true, subagent_session_id: null } }) + render() + await userEvent.click(screen.getByRole('button', { name: /Bash/i })) + expect(screen.getByText('boom')).toBeInTheDocument() + expect(screen.getByTestId('tool-error')).toBeInTheDocument() + }) + + it('offers a subagent link and fires the callback', async () => { + const onOpen = vi.fn() + const block = tool({ name: 'Task', input: { subagent_type: 'Explore', prompt: 'find x' }, + result: { content: 'done', is_error: false, subagent_session_id: 'child-9' } }) + render() + await userEvent.click(screen.getByRole('button', { name: /view subagent/i })) + expect(onOpen).toHaveBeenCalledWith('child-9') + }) +}) +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `npm run test -- ToolBlock.test.tsx` +Expected: FAIL (module not found). + +- [ ] **Step 3: Write minimal implementation** + +```tsx +// webapp/src/components/transcript/ToolBlock.tsx +import { ChevronRight, CornerDownRight, Wrench } from 'lucide-react' +import { useState } from 'react' +import type { TranscriptBlock } from '../../lib/transcript' +import { CodeBlock } from './CodeBlock' +import { DiffView } from './DiffView' + +type ToolUse = Extract + +/** One-line summary of the tool's input, agentsview-style. */ +function headline(block: ToolUse): string { + const i = block.input as Record + const s = (k: string) => (typeof i[k] === 'string' ? (i[k] as string) : '') + switch (block.name) { + case 'Bash': + case 'run_command': + return s('command') || s('cmd') + case 'Read': + case 'Edit': + case 'MultiEdit': + case 'Write': + case 'Update': + case 'NotebookEdit': + return s('file_path') || s('path') || s('notebook_path') + case 'Grep': + return s('pattern') + case 'Glob': + return s('pattern') || s('glob') + case 'Task': + case 'Agent': + return s('subagent_type') || s('description') || s('prompt').slice(0, 80) + default: + return '' + } +} + +function ResultBody({ block }: { block: ToolUse }) { + const r = block.result + if (!r) return

no output captured

+ const isEdit = ['Edit', 'MultiEdit', 'Write', 'Update'].includes(block.name) + if (isEdit && /^@@|\n[+-]/.test(r.content)) return + if (r.is_error) { + return ( +
+        {r.content}
+      
+ ) + } + return +} + +export function ToolBlock({ + block, + onOpenSubagent, +}: { + block: ToolUse + onOpenSubagent?: (sessionId: string) => void +}) { + const [open, setOpen] = useState(false) + const head = headline(block) + const child = block.result?.subagent_session_id ?? null + return ( +
+ + {open && ( +
+ {Object.keys(block.input).length > 0 && ( +
+ input + +
+ )} + + {child && onOpenSubagent && ( + + )} +
+ )} +
+ ) +} +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `npm run test -- ToolBlock.test.tsx` +Expected: PASS (4 passed). + +- [ ] **Step 5: Commit** + +```bash +git add webapp/src/components/transcript/ToolBlock.tsx webapp/src/components/transcript/ToolBlock.test.tsx +git commit -m "feat(webapp): tool block with per-tool rendering and diffs" +``` + +--- + +### Task 8: MessageBlock + TranscriptView + SessionsView + route/nav + +**Files:** +- Create: `webapp/src/components/transcript/MessageBlock.tsx` +- Create: `webapp/src/views/TranscriptView.tsx` +- Create: `webapp/src/views/SessionsView.tsx` +- Modify: `webapp/src/App.tsx` (add route) +- Modify: `webapp/src/components/Shell.tsx` (add nav item + title) +- Test: `webapp/src/views/SessionsView.test.tsx` + +**Interfaces:** +- Consumes: `ThinkingBlock`, `ToolBlock`, `Markdown`, `fetchTranscript`, `useConnection`, `useFanout`, `SessionEntry`. +- Produces: ``; ``; ``. + +- [ ] **Step 1: Write the failing test** + +```tsx +// webapp/src/views/SessionsView.test.tsx +import { screen, waitFor } from '@testing-library/react' +import userEvent from '@testing-library/user-event' +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('../lib/rpc', async () => { + const actual = await vi.importActual('../lib/rpc') + return { ...actual, rpc: vi.fn(), fetchHealth: vi.fn(), fetchCapabilities: vi.fn() } +}) +import { fetchCapabilities, fetchHealth, rpc } from '../lib/rpc' +import { renderWithProviders, seedConnection } from '../test/utils' +import { SessionsView } from './SessionsView' + +const CAPS = { name: 'vouch', version: '1', level: 3, methods: ['kb.list_sessions', 'kb.session_transcript'], review_gated: true } + +beforeEach(() => { + localStorage.clear() + vi.clearAllMocks() + vi.mocked(fetchHealth).mockResolvedValue(true) + vi.mocked(fetchCapabilities).mockResolvedValue(CAPS as never) + seedConnection() +}) + +describe('SessionsView', () => { + it('lists sessions and renders a picked transcript', async () => { + vi.mocked(rpc).mockImplementation(async (_c, method) => { + if (method === 'kb.list_sessions') { + return { sessions: [{ session_id: 'sid-1', stage: 'buffer', proposal_id: null, kind: null, + title: 'Fix parser', summarized: false, observations: 3, last_activity: '2026-07-10T00:00:00Z' }] } + } + if (method === 'kb.session_transcript') { + return { available: true, source: { agent: 'claude', path: '/x' }, + session: { id: 'sid-1', agent: 'claude', cwd: '/repo', git_branch: 'main', title: 'Fix parser', + started_at: null, ended_at: null, model: 'claude-opus-4-8', + tokens: { input: 1, output: 1, cache_read: 0, cache_creation: 0 } }, + messages: [{ role: 'assistant', id: 'm1', model: 'claude-opus-4-8', timestamp: null, tokens: null, + blocks: [{ type: 'text', text: 'hello from claude' }] }], truncated: false } + } + return {} + }) + renderWithProviders(, { route: '/sessions' }) + await waitFor(() => expect(screen.getByText('Fix parser')).toBeInTheDocument()) + await userEvent.click(screen.getByText('Fix parser')) + await waitFor(() => expect(screen.getByText('hello from claude')).toBeInTheDocument()) + }) +}) +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `npm run test -- SessionsView.test.tsx` +Expected: FAIL (module not found). + +- [ ] **Step 3: Write minimal implementation** + +```tsx +// webapp/src/components/transcript/MessageBlock.tsx +import { Bot, User } from 'lucide-react' +import { Markdown } from '../Markdown' +import type { TranscriptMessage } from '../../lib/transcript' +import { ThinkingBlock } from './ThinkingBlock' +import { ToolBlock } from './ToolBlock' + +export function MessageBlock({ + message, + onOpenSubagent, +}: { + message: TranscriptMessage + onOpenSubagent?: (sessionId: string) => void +}) { + const isUser = message.role === 'user' + return ( +
+
+ {isUser ? : } + {isUser ? 'user' : 'assistant'} + {message.model && {message.model}} +
+
+ {message.blocks.map((b, i) => { + if (b.type === 'thinking') return + if (b.type === 'tool_use') return + return ( +
+ {b.text} +
+ ) + })} +
+
+ ) +} +``` + +```tsx +// webapp/src/views/TranscriptView.tsx +import { useQuery } from '@tanstack/react-query' +import { useState } from 'react' +import { EmptyState } from '../components/EmptyState' +import { ErrorCard } from '../components/ErrorCard' +import { MessageBlock } from '../components/transcript/MessageBlock' +import { fetchTranscript } from '../lib/transcript' +import type { Observation } from '../lib/transcript' +import type { VouchConnectionInfo } from '../lib/types' +import { VouchRpcError } from '../lib/rpc' + +function Degraded({ reason, observations }: { reason: string; observations: Observation[] }) { + return ( +
+
+ original transcript unavailable — {reason}. Showing captured activity. +
+ {observations.length === 0 ? ( + + ) : ( +
    + {observations.map((o, i) => ( +
  1. + {o.tool} + {o.summary} +
  2. + ))} +
+ )} +
+ ) +} + +export function TranscriptView({ + conn, + sessionId, + agent, +}: { + conn: VouchConnectionInfo + sessionId: string + agent?: string +}) { + // Subagent drill-down replaces the shown transcript with the child's, with a back stack. + const [stack, setStack] = useState<{ id: string; agent?: string }[]>([{ id: sessionId, agent }]) + const top = stack[stack.length - 1] + const q = useQuery({ + queryKey: ['transcript', conn.endpoint, top.id], + queryFn: () => fetchTranscript(conn, top.id, top.agent), + }) + + if (q.isPending) return
Loading transcript…
+ if (q.isError) { + const e = q.error + return
+ } + const t = q.data + return ( +
+ {stack.length > 1 && ( + + )} + {!t.available ? ( + + ) : ( + <> +
+ {t.session.model && {t.session.model}} + {t.session.cwd && {t.session.cwd}} + {t.session.git_branch && ⎇ {t.session.git_branch}} + {t.session.tokens.input + t.session.tokens.output} tokens + {t.source.agent} +
+ {t.truncated && ( +
+ transcript truncated at {t.messages.length} messages +
+ )} + {t.messages.map((m, i) => ( + setStack((s) => [...s, { id, agent: t.source.agent }])} /> + ))} + + )} +
+ ) +} +``` + +```tsx +// webapp/src/views/SessionsView.tsx +import { useState } from 'react' +import { EmptyState } from '../components/EmptyState' +import { useConnection } from '../connection/ConnectionContext' +import { useFanout } from '../lib/fanout' +import type { SessionEntry } from '../lib/types' +import type { VouchConnectionInfo } from '../lib/types' +import { TranscriptView } from './TranscriptView' + +interface Row { conn: VouchConnectionInfo; label: string; s: SessionEntry } + +export function SessionsView() { + const { hasMethod } = useConnection() + const sessions = useFanout<{ sessions: SessionEntry[] }>(['sessions'], 'kb.list_sessions', {}, { + refetchInterval: 10_000, + }) + const rows: Row[] = sessions.rows.flatMap((r) => + (r.data?.sessions ?? []).map((s) => ({ conn: r.project.conn, label: r.project.label, s })), + ) + const [sel, setSel] = useState(null) + + return ( +
+ +
+ {sel && sel.s.session_id ? ( + + ) : ( +
+ )} +
+
+ ) +} +``` + +In `webapp/src/App.tsx`, add the import and route (after the `/stats` route): + +```tsx +import { SessionsView } from './views/SessionsView' +// ... + } /> +``` + +In `webapp/src/components/Shell.tsx`, import an icon and add nav + title: + +```tsx +// add ScrollText to the lucide-react import +import { Activity, BadgeCheck, FileClock, Inbox, Library, MessageSquare, Plug, ScrollText, SunMoon } from 'lucide-react' +// add to NAV array (before /stats): + { to: '/sessions', label: 'Sessions', icon: ScrollText }, +// add to TITLES: + '/sessions': 'Session transcripts', +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `npm run test -- SessionsView.test.tsx` +Expected: PASS. + +- [ ] **Step 5: Full frontend gates** + +Run: `npm run test && npm run build` +Expected: all tests pass; `tsc && vite build` succeeds with no type errors. + +- [ ] **Step 6: Commit** + +```bash +git add webapp/src/components/transcript/MessageBlock.tsx webapp/src/views/TranscriptView.tsx webapp/src/views/SessionsView.tsx webapp/src/views/SessionsView.test.tsx webapp/src/App.tsx webapp/src/components/Shell.tsx +git commit -m "feat(webapp): sessions tab with full transcript viewer" +``` + +--- + +## PHASE 3 — Codex, subagents polish, e2e + +### Task 9: Codex full-transcript parser + +**Files:** +- Modify: `src/vouch/transcript.py` (replace the `parse_codex_transcript` stub) +- Test: `tests/test_session_transcript.py` + +**Interfaces:** +- Consumes: `codex_rollout._iter_rollout_lines` is NOT reused (it raises on zstd); parse plain records directly. +- Produces: `parse_codex_transcript(path, *, max_messages=2000) -> dict[str, Any]` returning the same schema, `session.agent == "codex"`. + +Codex rollout records (verified in `codex_rollout.py`): `{"type":"session_meta","payload":{"id","cwd","timestamp"}}`; `{"type":"event_msg","payload":{"type":"user_message","message":"…"}}` and `agent_message` (assistant text, field `message`); `{"type":"response_item","payload":{"type":"function_call","name","arguments","call_id"}}` and `{"type":"response_item","payload":{"type":"function_call_output","call_id","output"}}`. + +- [ ] **Step 1: Write the failing test** + +```python +# append to tests/test_session_transcript.py +_CODEX_LINES = [ + {"type": "session_meta", "payload": {"id": "cx-1", "cwd": "/repo", "timestamp": "2026-06-22T08:01:54Z"}}, + {"type": "event_msg", "payload": {"type": "user_message", "message": "run the tests"}}, + {"type": "event_msg", "payload": {"type": "agent_message", "message": "Running them now."}}, + {"type": "response_item", "payload": {"type": "function_call", "name": "shell", + "arguments": "{\"command\": \"pytest\"}", "call_id": "c1"}}, + {"type": "response_item", "payload": {"type": "function_call_output", "call_id": "c1", + "output": "1 passed"}}, +] + + +def test_parse_codex_pairs_calls(tmp_path: Path) -> None: + f = tmp_path / "rollout-x.jsonl" + _write_jsonl(f, _CODEX_LINES) + out = transcript.parse_codex_transcript(f) + assert out["session"]["agent"] == "codex" + assert out["session"]["cwd"] == "/repo" + roles = [m["role"] for m in out["messages"]] + assert roles[0] == "user" + # the assistant message carries the agent text and the paired tool call + assistant = next(m for m in out["messages"] if m["role"] == "assistant") + types = [b["type"] for b in assistant["blocks"]] + assert "tool_use" in types and "text" in types + tu = next(b for b in assistant["blocks"] if b["type"] == "tool_use") + assert tu["name"] == "shell" + assert tu["result"]["content"] == "1 passed" +``` + +- [ ] **Step 2: Run test to verify it fails** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k codex -q` +Expected: FAIL — `NotImplementedError` from the stub. + +- [ ] **Step 3: Write minimal implementation** (replace the stub) + +```python +def parse_codex_transcript(path: Path, *, max_messages: int = 2000) -> dict[str, Any]: + """Parse a Codex rollout JSONL into the normalized transcript schema. + + Codex interleaves user/agent messages (``event_msg``) with tool calls + and outputs (``response_item``). Consecutive assistant activity — agent + text and its function calls — is grouped into one assistant message; + a ``function_call_output`` pairs into its call by ``call_id``. + """ + session: dict[str, Any] = { + "id": path.stem, "agent": "codex", "cwd": None, "git_branch": None, + "title": None, "started_at": None, "ended_at": None, "model": None, + "tokens": {"input": 0, "output": 0, "cache_read": 0, "cache_creation": 0}, + } + messages: list[dict[str, Any]] = [] + tool_by_call: dict[str, dict[str, Any]] = {} + truncated = False + current: dict[str, Any] | None = None + + def new_assistant() -> dict[str, Any]: + return {"role": "assistant", "id": None, "model": None, "timestamp": None, + "tokens": None, "blocks": []} + + with path.open(encoding="utf-8") as fh: + for raw in fh: + raw = raw.strip() + if not raw: + continue + try: + rec = json.loads(raw) + except json.JSONDecodeError: + continue + if not isinstance(rec, dict): + continue + payload = rec.get("payload") + if not isinstance(payload, dict): + continue + rtype = rec.get("type") + if len(messages) >= max_messages: + truncated = True + break + + if rtype == "session_meta": + if isinstance(payload.get("id"), str) and session["id"] == path.stem: + session["id"] = payload["id"] + if isinstance(payload.get("cwd"), str): + session["cwd"] = payload["cwd"] + if isinstance(payload.get("timestamp"), str): + session["started_at"] = payload["timestamp"] + session["ended_at"] = payload["timestamp"] + elif rtype == "event_msg" and payload.get("type") == "user_message": + if current is not None: + messages.append(current) + current = None + text = str(payload.get("message", "")).strip() + if text: + messages.append({"role": "user", "id": None, "model": None, + "timestamp": None, "tokens": None, + "blocks": [{"type": "text", "text": text}]}) + elif rtype == "event_msg" and payload.get("type") == "agent_message": + if current is None: + current = new_assistant() + text = str(payload.get("message", "")).strip() + if text: + current["blocks"].append({"type": "text", "text": text}) + elif rtype == "response_item" and payload.get("type") == "function_call": + if current is None: + current = new_assistant() + block = {"type": "tool_use", "id": payload.get("call_id"), + "name": str(payload.get("name", "")), + "input": {"arguments": payload.get("arguments")}, "result": None} + current["blocks"].append(block) + cid = payload.get("call_id") + if isinstance(cid, str): + tool_by_call[cid] = block + elif rtype == "response_item" and payload.get("type") == "function_call_output": + cid = payload.get("call_id") + block = tool_by_call.get(cid) if isinstance(cid, str) else None + if block is not None: + block["result"] = {"content": str(payload.get("output", "")), + "is_error": False, "subagent_session_id": None} + if current is not None: + messages.append(current) + return {"session": session, "messages": messages, "truncated": truncated} +``` + +- [ ] **Step 4: Run test to verify it passes** + +Run: `.venv/bin/python -m pytest tests/test_session_transcript.py -k codex -q` +Expected: PASS. + +- [ ] **Step 5: Full backend gates + commit** + +```bash +.venv/bin/python -m pytest tests/ -q --ignore=tests/embeddings && .venv/bin/python -m mypy src && .venv/bin/python -m ruff check src tests +git add src/vouch/transcript.py tests/test_session_transcript.py +git commit -m "feat(transcript): parse Codex rollouts into normalized blocks" +``` + +--- + +### Task 10: Degraded-render + subagent test coverage (frontend) + +**Files:** +- Create: `webapp/src/views/TranscriptView.test.tsx` + +**Interfaces:** consumes existing `TranscriptView`. + +- [ ] **Step 1: Write the failing test** + +```tsx +// webapp/src/views/TranscriptView.test.tsx +import { screen, waitFor } from '@testing-library/react' +import userEvent from '@testing-library/user-event' +import { beforeEach, describe, expect, it, vi } from 'vitest' + +vi.mock('../lib/rpc', async () => { + const actual = await vi.importActual('../lib/rpc') + return { ...actual, rpc: vi.fn(), fetchHealth: vi.fn(), fetchCapabilities: vi.fn() } +}) +import { rpc } from '../lib/rpc' +import { renderWithProviders } from '../test/utils' +import { TranscriptView } from './TranscriptView' + +const conn = { endpoint: 'http://127.0.0.1:8731' } + +beforeEach(() => { vi.clearAllMocks() }) + +describe('TranscriptView', () => { + it('renders the degraded observation timeline', async () => { + vi.mocked(rpc).mockResolvedValue({ available: false, reason: 'raw transcript not found', + observations: [{ ts: 1, tool: 'Edit', summary: 'Edited types.go' }] }) + renderWithProviders() + await waitFor(() => expect(screen.getByText('Edited types.go')).toBeInTheDocument()) + expect(screen.getByText(/original transcript unavailable/i)).toBeInTheDocument() + }) + + it('drills into a subagent and back', async () => { + vi.mocked(rpc).mockImplementation(async (_c, _m, params) => { + const id = (params as { session_id: string }).session_id + const text = id === 'child-9' ? 'child says hi' : 'parent turn' + const blocks = id === 'child-9' + ? [{ type: 'text', text }] + : [{ type: 'tool_use', id: 't1', name: 'Task', input: { prompt: 'go' }, + result: { content: 'done', is_error: false, subagent_session_id: 'child-9' } }] + return { available: true, source: { agent: 'claude', path: '/x' }, + session: { id, agent: 'claude', cwd: null, git_branch: null, title: null, started_at: null, + ended_at: null, model: null, tokens: { input: 0, output: 0, cache_read: 0, cache_creation: 0 } }, + messages: [{ role: 'assistant', id: 'm', model: null, timestamp: null, tokens: null, blocks }], + truncated: false } + }) + renderWithProviders() + await waitFor(() => expect(screen.getByText('Task')).toBeInTheDocument()) + await userEvent.click(screen.getByRole('button', { name: /Task/i })) + await userEvent.click(screen.getByRole('button', { name: /view subagent/i })) + await waitFor(() => expect(screen.getByText('child says hi')).toBeInTheDocument()) + await userEvent.click(screen.getByRole('button', { name: /back to parent/i })) + await waitFor(() => expect(screen.getByText('Task')).toBeInTheDocument()) + }) +}) +``` + +- [ ] **Step 2: Run test to verify it fails, then passes** + +Run: `npm run test -- TranscriptView.test.tsx` +Expected: These should PASS against the Task 8 implementation. If the subagent-back or degraded rendering fails, fix `TranscriptView` (this task is the regression net that proves both paths). Do not modify tests to fit bugs. + +- [ ] **Step 3: Full frontend gates + commit** + +```bash +npm run test && npm run build +git add webapp/src/views/TranscriptView.test.tsx +git commit -m "test(webapp): cover degraded render and subagent drill-down" +``` + +--- + +### Task 11: E2E smoke + +**Files:** +- Create: `webapp/e2e/sessions.spec.ts` + +**Interfaces:** exercises the running app against the existing e2e fixture server (see `webapp/e2e/global-setup.ts`). + +- [ ] **Step 1: Inspect the existing e2e harness** + +Run: `sed -n '1,60p' webapp/e2e/smoke.spec.ts webapp/e2e/global-setup.ts` +Learn how the fixture endpoint is seeded and how `kb.*` responses are stubbed/served, then mirror that to stub `kb.list_sessions` + `kb.session_transcript`. (The spec must drive the real UI — navigate to Sessions, click a row, assert a rendered block — not assert on source.) + +- [ ] **Step 2: Write the spec** + +```ts +// webapp/e2e/sessions.spec.ts +import { expect, test } from '@playwright/test' + +// Follow the pattern established in smoke.spec.ts for seeding a connection +// and stubbing /proxy/rpc. Route kb.list_sessions -> one row with a +// session_id, and kb.session_transcript -> an available transcript with a +// single assistant text block "e2e transcript body". +test('sessions tab renders a picked transcript', async ({ page }) => { + // ...seed + route stubs mirroring smoke.spec.ts... + await page.goto('/sessions') + await page.getByText('e2e session').click() + await expect(page.getByText('e2e transcript body')).toBeVisible() +}) +``` + +- [ ] **Step 3: Run it** + +Run: `npm run e2e -- sessions.spec.ts` +Expected: PASS. (If the harness cannot stub per-method easily, assert the empty-state path instead: navigating to `/sessions` shows "No sessions" — still real behavior, no tautology.) + +- [ ] **Step 4: Commit** + +```bash +git add webapp/e2e/sessions.spec.ts +git commit -m "test(webapp): e2e smoke for the sessions transcript tab" +``` + +--- + +### Task 12: Spec sync + docs + +**Files:** +- Modify: `docs/superpowers/specs/2026-07-10-session-transcript-viewer-design.md` + +- [ ] **Step 1:** Update the spec's "Backend contract" to reflect server-side tool_result pairing (blocks carry `result` inline; no separate `tool_result` block) and resolve the open questions to what shipped (Sessions tab, master–detail, Codex in v1). Run `mdformat --wrap 80 docs/superpowers/specs/2026-07-10-session-transcript-viewer-design.md` if available. + +- [ ] **Step 2: Commit** + +```bash +git add docs/superpowers/specs/2026-07-10-session-transcript-viewer-design.md +git commit -m "docs(transcript): align spec with shipped contract" +``` + +--- + +## Self-Review + +**Spec coverage:** Backend RPC + `transcript.py` (Tasks 1-4, 9). Locator Claude + Codex (1, 9). Normalized schema (2, 9). Degradation to observations (3, 10). Frontend SessionsView + TranscriptView + block renderers (5-8). Per-tool rendering + diffs (7). Subagent lazy expansion (7, 8, 10). Capability gating (8). Tests every task; e2e (11). Conventions/guardrails in Global Constraints. All spec sections map to a task. + +**Placeholder scan:** No TBD/TODO. The only forward reference is `parse_codex_transcript` (stub in Task 3, implemented Task 9) — explicitly flagged with a raising stub so Phase 1 stays green. E2E spec body references the existing harness pattern (Task 11 Step 1 reads it first) rather than inventing an unknown stub API. + +**Type consistency:** Block schema identical across backend (dicts) and `lib/transcript.ts` (`TranscriptBlock`). `tool_use.result` is `{content, is_error, subagent_session_id}` everywhere. `fetchTranscript(conn, sessionId, agent?)`, `load_transcript(store, session_id, *, agent=None)`, and the handler param names (`session_id`, `agent`) match. `useFanout` row shape (`{project, data}`) matches Task 8 usage. + +--- + +## Execution Handoff + +Recommended: **Subagent-Driven** (fresh subagent per task, two-stage review between tasks). Phase 1 (backend, Tasks 1-4) is the critical path and must be green before Phase 2 consumes the RPC. Alternative: **Inline Execution** with checkpoints after each phase. diff --git a/docs/superpowers/specs/2026-07-09-artifact-delete-design.md b/docs/superpowers/specs/2026-07-09-artifact-delete-design.md new file mode 100644 index 00000000..692eebfd --- /dev/null +++ b/docs/superpowers/specs/2026-07-09-artifact-delete-design.md @@ -0,0 +1,239 @@ +# review-gated artifact delete + +status: draft +date: 2026-07-09 +scope: add a hard-delete path for durable artifacts (claim, page, entity, +relation) that routes through the review gate. + +## why + +vouch has a soft-delete today — `kb.archive` flips a claim to +`status=archived`, keeps the file and every link. there is no way to +*remove* an artifact. junk that should never have been written (a +mis-slugged claim, a duplicate entity, an auto-extracted edge that is +plain wrong) can only be hidden, never erased. this leaves the on-disk +kb and its diffs cluttered with records the maintainer has explicitly +judged to be garbage. + +hard delete fills that gap. it is deliberately the most consequential +write in the system, so it is held to the strongest control vouch has: +the review gate. + +## decisions (locked) + +1. **hard delete.** the artifact's file is physically removed from + `.vouch/`, dropped from the derived index, and the audit event is the + only in-place trace. git history still holds the file — delete is not + "unrecoverable", it is "gone from the working kb". +2. **through the review gate.** deletion is a proposal. an agent (or + human) files `kb.propose_delete`; a *different* reviewer approves it + via the existing `kb.approve`. no direct-mutation path exists, in + keeping with the north star: every write is reviewed, and no parallel + path bypasses `proposals.approve()`. +3. **block if referenced.** a delete is refused — at propose time as a + friendly error, and again at approve time as the authoritative gate — + if any other artifact points at the target. the maintainer must + supersede or remove the referring artifacts first. this preserves + vouch's existing no-dangling-refs invariant (the same invariant the + `put_*` write guards and `kb.lint`/`kb.doctor` enforce). +4. **all four durable kinds.** claim, page, entity, and relation are all + deletable through one generic path. (sources and evidence are out of + scope for this pass — see "out of scope".) + +## object model + +a delete is represented as a new proposal kind, not smuggled through an +existing one. + +* `ProposalKind.DELETE = "delete"` (new enum member in `models.py`). +* payload shape: + + ```yaml + target_kind: claim | page | entity | relation + id: + snapshot: { ...full model_dump of the artifact at propose time... } + ``` + +`target_kind` is explicit rather than inferred: claim / page / entity +slugs can collide across kinds, so the caller must name the kind. the +`snapshot` is the full serialized artifact captured at propose time; it +makes the resulting `decided/.yaml` proposal a tombstone record of +exactly what was removed, and is copied into the audit event. the +snapshot is for the reviewer's context and the audit trail only — the +actual delete re-reads live state (see "approve"). + +## the reference matrix + +"referenced" means an *inbound* pointer from another artifact. outbound +refs (what the target itself points at) never block a delete: removing +the holder simply drops its own pointers, and the things it pointed at +survive. + +| deleting a… | blocked if referenced (inbound) by | storage remove | index remove | +|---|---|---|---| +| **claim** | page `claims[]`; relation `source`/`target`; another claim's `supersedes` / `superseded_by` / `contradicts` | unlink `claims/.yaml` | `claims_fts`, `embedding_index(claim,id)`, `prov_edges` touching id | +| **page** | relation `source`/`target` (pages are valid relation endpoints) | unlink `pages/.md` | `pages_fts`, `embedding_index(page,id)`, `prov_edges` touching id | +| **entity** | claim `entities[]`; page `entities[]`; relation `source`/`target` | unlink `entities/.yaml` | `entities_fts`, `embedding_index(entity,id)`, `prov_edges` touching id | +| **relation** | *nothing* — an edge has no inbound refs → always deletable | unlink `relations/.yaml` | `embedding_index(relation,id)`, `prov_edges` touching id (relations are embedded on write but never in FTS) | + +a consequence worth stating plainly: to delete a page or entity that a +relation points at, you first delete those inbound relations (relations +delete freely, having no inbound refs of their own). that chain is the +"block if referenced" contract behaving as designed — the maintainer +never silently orphans an edge. + +## flow + +### propose + +`proposals.propose_delete(store, *, target_kind, target_id, proposed_by, +rationale=None, session_id=None, dry_run=False) -> Proposal` + +1. validate `target_kind` is one of the four kinds. +2. resolve the artifact via the matching getter; raise `ProposalError` + if it does not exist. +3. run `referenced_by(store, target_kind, target_id)`. if it returns a + non-empty list, raise `ProposalError` naming the referrers. for a + claim, the message also points at `supersede` as the usual intended + operation. +4. snapshot = `artifact.model_dump(mode="json")`. +5. file via the existing `_file_proposal` with `kind=DELETE`. dry-run + behaves exactly as the other `propose_*` paths (no disk write, audit + `proposal.delete.dry_run`). + +### approve + +a new branch in `proposals.approve()`, dispatched on +`proposal.payload["target_kind"]`. + +* **skip `_ensure_no_existing_artifact`.** that guard refuses to write + over an existing artifact; a delete is the inverse — the artifact must + exist. the guard is currently applied to every non-page kind, so the + approve path is restructured to exempt DELETE. +* **re-resolve the artifact.** if it is already gone, treat the delete as + already done: skip the unlink, but still mark the proposal decided and + write the audit event. this makes approve idempotent under a + crash-retry between the unlink and `move_proposal_to_decided`. +* **re-run `referenced_by`.** refs may have appeared since propose time + (mirroring how page approval re-validates the page kind at the gate). + if now referenced, raise `ProposalError` — the delete is blocked, the + proposal stays pending. +* **remove.** call `storage.delete_(id)` then + `index_db.deindex(conn, kind, id)`. +* **audit.** write a per-kind `claim.delete` / `page.delete` / + `entity.delete` / `relation.delete` event carrying the snapshot, plus + the generic `proposal.delete.approve` the approve path already emits. +* mark the proposal `APPROVED` and move it to `decided/` exactly as every + other kind does. + +### reject / list / expire + +no new code. `kb.reject`, `kb.list_pending`, `kb.triage_pending`, and +`expire_pending` are all kind-agnostic — a rejected delete simply leaves +the target untouched, and a stale delete proposal expires like any other. + +### batch precheck + +`check_approvable` / `_payload_block_reason` gain a DELETE branch: verify +the target exists and is unreferenced, so `vouch approve a b` stays +all-or-nothing when a delete is in the batch. + +### shared helper + +`referenced_by(store, target_kind, target_id) -> list[str]` implements +the matrix. returns human-readable referrer descriptions (e.g. +`page 'foo'`, `relation x--rel--abc`). used by propose, approve, and +`check_approvable` so the three paths never drift. + +## storage layer + +`storage.py` gains four pure-I/O unlink methods mirroring the existing +source-unlink pattern — no business logic, no ref checks (those live in +`proposals`): + +* `delete_claim(claim_id)` → unlink `claims/.yaml` +* `delete_page(page_id)` → unlink `pages/.md` +* `delete_entity(entity_id)` → unlink `entities/.yaml` +* `delete_relation(relation_id)` → unlink `relations/.yaml` + +each raises `ArtifactNotFoundError` if the file is absent (the approve +path treats that as the idempotent already-deleted case). + +## index layer + +`index_db.py` gains `deindex(conn, *, kind, id)`: + +* `kind in {claim, page, entity}` → `DELETE FROM s_fts WHERE id=?` + (relations have no FTS table). +* all four kinds → `DELETE FROM embedding_index WHERE kind=? AND id=?`. + every `put_*` calls `_embed_and_store`, so an embedding row can exist + for any kind, relations included (present only when the embeddings + extra is installed; the delete is a harmless no-op otherwise). +* all kinds → remove `prov_edges` rows referencing the id (as `src_id` + or `dst_id`). `prov_edges` is otherwise derived and rebuildable via + `kb.provenance_rebuild`; removing the touched rows inline keeps + `state.db` consistent without a full rebuild. + +## surfaces (four registration sites + test) + +one new method, `kb.propose_delete`, mirrored across every surface per +the "when you add a new kb.\* method" checklist in CLAUDE.md: + +1. **MCP** — `kb_propose_delete(target_kind, target_id, rationale=None)` + in `server.py`. +2. **JSONL** — `_h_propose_delete` + `HANDLERS["kb.propose_delete"]` in + `jsonl_server.py`. +3. **METHODS** — `"kb.propose_delete"` in `capabilities.py`. +4. **CLI** — `vouch propose-delete ` in + `cli.py`, echoing the filed proposal id. sits alongside the existing + `supersede` / `archive` / `contradict` lifecycle commands. + +no new approve/reject method — the human uses the existing +`vouch approve ` / `kb.approve`. + +`test_capabilities` enforces method-list parity across all four; a new +`tests/test_delete.py` covers the behavior. + +## testing + +`tests/test_delete.py`: + +* propose + approve happy path for each of the four kinds; assert the + file is gone, the index row is gone, and the audit event is written. +* block-if-referenced for each kind per the matrix (claim cited by a + page; entity in a claim's `entities`; page as a relation endpoint; + etc.); assert `propose_delete` raises and nothing is written. +* the block is re-checked at approve: propose a valid delete, then add a + referring artifact, then approve → raises, target survives, proposal + stays pending. +* relation deletes freely even when its endpoints exist. +* idempotent approve: delete the underlying file out from under a pending + approved-in-flight proposal → approve still finalizes cleanly. +* forbidden self-approval still applies (proposer ≠ approver, unless + `review.approver_role: trusted-agent`). +* `check_approvable` returns the block reason for a referenced target so + the batch path is all-or-nothing. +* dry-run writes nothing and returns the would-be proposal id. + +the ci gate is unchanged: `pytest tests/ -q --ignore=tests/embeddings`, +`mypy src`, `ruff check src tests`. + +## out of scope + +* **source / evidence delete.** sources have their own directory shape + (`_source_dir`, a directory not a single file) and different reference + semantics (claims cite them). deferred; the generic `target_kind` + dispatch leaves room to add them without reshaping the payload. +* **cascade delete.** explicitly rejected in favor of "block if + referenced" — one approval never mutates a second reviewed artifact. +* **undo / restore.** git history is the recovery path; the snapshot in + `decided/` is the record. no in-product un-delete in this pass. +* **web review-ui rendering.** the review console lives in the separate + vouch-ui repo. it will need a row renderer for the DELETE kind (show + "delete " with the snapshot); tracked there, not built here. + +## north-star check + +deletion is a proposal approved through `proposals.approve()`. there is +no direct mutation and no parallel data path. the review gate remains the +single chokepoint for every write, destructive ones included. diff --git a/docs/superpowers/specs/2026-07-09-session-split-summaries-design.md b/docs/superpowers/specs/2026-07-09-session-split-summaries-design.md new file mode 100644 index 00000000..875f5d2b --- /dev/null +++ b/docs/superpowers/specs/2026-07-09-session-split-summaries-design.md @@ -0,0 +1,296 @@ +# split a huge session into separate topical pages, host-neutrally + +- status: draft, awaiting review +- date: 2026-07-09 +- scope: one implementation plan + +## goal + +when a session is large enough that a single rolled-up summary would be an +unreadable wall, summarize it *with an llm* into several coherent topical +pages instead of one — one page per thread of work. this must work the same +for any host that feeds vouch (claude code, openclaw, codex, cursor, cline, +continue, windsurf, zed, …), not just claude code. small sessions keep today's +mechanical single-page rollup untouched, and every produced page is still a +`PENDING` proposal a human approves. + +## north-star fit + +vouch's load-bearing invariant is "every write goes through a review gate." +this feature adds an llm *drafting* path but keeps the *write* path gated: +each topical page lands as a `PENDING` page proposal via +`proposals.propose_page`, exactly like a hand-filed write or a compiled wiki +page. nothing is auto-approved; `approve()` is never called. the llm decides +how to slice the session, code verifies the slices mechanically, and the human +review is the gate. + +it also respects the second invariant the codebase already encodes: session +records are *raw material*, not durable wiki topic pages. the split pages stay +`type: session` — the same feedstock kind `capture.finalize` emits today, just +several topical ones instead of one monolith. `compile.py` remains the sole +citation-verified path from approved claims to durable topic pages, and it +still forbids `session`/`log` as output types. this feature does not erode that +boundary. + +## background: the load-bearing constraint + +vouch is going multi-host. twelve adapters already exist under `adapters/` +(claude-code, claude-desktop, cline, codex, continue, cursor, generic-mcp, +http-tunnel, jsonl-shell, openclaw, windsurf, zed). the summarizer therefore +cannot be written against claude code's session-end hook, its tool vocabulary, +or its transcript format. + +the seam that makes host-neutrality possible already exists: the **observation +buffer**. every host normalizes its native activity into one compact, +host-agnostic shape — `{ts, tool, summary, files?, cmd?}` — written via +`capture.observe` into `.vouch/captures/.jsonl`: + +- claude code maps `PostToolUse` payloads via `capture.summarize_tool`. +- codex parses rollout files via `codex_rollout.py` into "the same observation + shape `capture.observe` produces" and reuses `build_summary_body`. +- openclaw feeds observations through its context-engine ingest. +- mcp-only hosts write observations through the same `observe` contract. + +so "summarize a huge session" reads *only* this normalized buffer. the +summarizer never touches a transcript and never assumes a tool name. ingestion +is host-specific and thin; summarization is host-blind. + +### the three leaks in the claude-code-first draft (rejected) + +the first draft of this design leaked host assumptions in three places, all +removed here: + +1. it triggered off claude code's `SessionEnd` hook. → triggers now converge on + one neutral entry that every host reaches its own way. +2. it pulled the page title from `capture.first_user_prompt`, which parses + claude code's transcript jsonl. → the session intent now comes from a + neutral, prioritized source; transcript parsing is a per-host enricher, not + a core dependency. +3. it framed the whole feature as a `capture.py` (claude-code-flavored) + concern. → the summarizer is extracted into its own host-blind module. + +## design + +### architecture + +``` + HOST EDGE (per-host, thin) NEUTRAL CORE (host-blind) + ──────────────────────── ───────────────────────── + claude code PostToolUse ─┐ + codex rollout ─────┤ observe() ┌─ .vouch/captures/.jsonl + openclaw ingest ──────┼──────────────►│ {ts,tool,summary,files,cmd} + cursor/cline/… mcp ───────┘ └─ + optional intent header + │ + ▼ + session_split.summarize(id) + size-gate → mechanical | llm split + │ + ▼ + N PENDING type:session proposals +``` + +### new module: `src/vouch/session_split.py` (host-blind) + +one public entry: + +```python +summarize(store, session_id, *, intent=None, cwd=None, project=None, + generated_at=None, mode="auto", config=None) -> dict +``` + +reads the buffer observations + a git-diff backstop, applies the size gate, +runs the mechanical rollup *or* the llm topical split, files `PENDING` +proposals, deletes the buffer, and writes an audit event. returns +`{captured, summary_proposal_ids, summary_proposal_id, mode, dropped, truncated}`. + +`mode` is `"auto"` (gate decides), `"split"` (force llm), or `"mechanical"` +(force the single page). `summary_proposal_id` (first id or `None`) is retained +for backward compatibility with existing `finalize` callers; `summary_proposal_ids` +is the full list. + +### new module: `src/vouch/llm_draft.py` (extracted from compile.py) + +`run_llm(cmd, prompt, *, timeout_seconds)` + `parse_drafts(raw)` + fence +stripping, run in a throwaway temp cwd with forced utf-8 on both pipe +directions (the locale is latin-1 on some hosts). `compile.py` and +`session_split.py` both import it; `compile.py` keeps its claim-citation +validation. this is a targeted dedup of code both callers need, not a +speculative abstraction. + +### `src/vouch/capture.py` (trimmed to ingestion) + +keeps `observe`, `summarize_tool`, the buffer helpers, `finalize_all_except`, +and `build_summary_body` (the mechanical summarizer, reused by both branches). +`finalize()` stays as the claude-code / codex-facing wrapper: it resolves the +claude-code transcript intent (its one host-specific enricher), then delegates +to `session_split.summarize`. + +### control flow: the three-tier size gate + +``` +observations + git-diff → total = len(obs) + len(changed_files) + + total < min_observations ............... delete buffer, no page (unchanged) + mode=auto, total < threshold ........... mechanical single page (unchanged) + mode=auto, total ≥ threshold, llm ok ... llm topical split → N pages (new) + any llm failure / no llm_cmd ........... mechanical single page (fallback) +``` + +`threshold_observations` (default 40) sits above `min_observations` (default +3), so small sessions are never handed to an llm. the mechanical rollup is both +the default for mid-size sessions and the universal fallback. + +### the split prompt & output contract (host-neutral) + +role: "session historian." inlined input: + +- the resolved **intent** string, if any +- observation lines rendered `- [] (files: …)`, where `` + is an opaque label (no claude-code tool-name enum) so any host's vocabulary + clusters equally +- the git stat, if any +- **taken topics**: existing + pending page names, the same dedupe list + `compile.py` builds + +rules given to the llm: cluster into at most `max_pages` coherent *topics*, one +page each; each page `{title, body}` with an 80–200-word markdown body and a +specific title ("fixed the audit-log write race", not "bug fixes"); **no +`[claim: id]` markers** — their absence is what marks these as uncited +feedstock, distinct from compile's cited pages; output *only* a json array of +`{title, body}` objects, no prose, no code fences. + +### validation (mechanical, per draft) + +drop-with-reason when: title or body is empty; the title slug collides with an +existing, pending, or in-batch page (the same overwrite-on-approve guard +`compile.py` uses, since `approve()` routes a colliding id through +`update_page`); or the draft is past the `max_pages` cap (cap is first-come, so +the outcome does not depend on drop order). `type` is **forced to `session`** in +code regardless of what the llm emits. survivors are filed: + +```python +propose_page(store, title=…, body=…, page_type="session", + tags=["session", "split"], session_id=session_id, + metadata={"session_id": session_id}, + proposed_by="session-split", + rationale="llm topical split of session ") +``` + +`approve()` is never called. the producing agent is already recorded on the +`Session` (`Session.agent`, set at `session_start`) and on the proposal's +`session_id`, so no separate host field is stored on the page. + +### "huge" input handling + +v1 is a single llm call feeding the whole compact buffer. observation records +are ~30–40 tokens each, so a realistic session — even a very long one — fits a +modern context. guard: if the serialized prompt would exceed `max_input_chars` +(default ~60 000), keep the most-recent observations that fit, append an +explicit `(N older observations elided)` note to the prompt, set `truncated` in +the result, and log it. **no silent cap.** map-reduce chunking (chunk → per-chunk +topic candidates → merge) is documented as a phase-2 upgrade if buffers routinely +exceed the budget; it is deliberately out of scope for v1 to avoid spending two +llm calls on every large session. + +### intent resolution (neutral priority order) + +1. `Session.task` — the neutral field set at `session_start` +2. an `intent` header record the adapter may write into the buffer +3. a host-registered transcript parser (claude code registers its + `first_user_prompt`; other hosts need not) +4. a filename/observation-derived fallback + +the summarizer core never parses a transcript itself; step 3 is the only seam +where host-specific parsing may run, and it is optional. + +### error handling + +no resolvable `llm_cmd`, a nonzero exit, a timeout, non-json output, or zero +valid drafts after validation → warn + fall back to the mechanical single page +→ one `PENDING` proposal, `mode="fallback"`. `summarize()` never raises to a +hook and never leaves a session unsummarized. the buffer is deleted only after +a page is filed (or an explicit below-min skip), so a crash mid-run leaves the +buffer intact for the next `finalize-all` sweep to retry. + +### audit + +new event `session.split`, actor = the triggering human/token label (or the +`session-split` proposer): `object_ids` = the filed proposal ids, `data = +{mode, proposed, dropped, observations, truncated}`. mirrors `compile.run` so +host-triggered splits stay attributable. + +### surfaces + +- **`kb.summarize_session`** — new neutral method (`session_id`, `mode`), the + trigger for hosts that cannot run shell hooks. four registration sites the + `test_capabilities` parity check enforces: mcp tool in `server.py`, jsonl + handler + `HANDLERS` entry in `jsonl_server.py`, the `METHODS` list in + `capabilities.py`, and a cli command. write-only, so it attaches no + `_meta.vouch_salience` sidebar. +- **cli**: `vouch capture finalize` gains `--split/--no-split` (maps to `mode`); + plus `vouch capture summarize [--split/--no-split]`. + +### triggers (per-host, one convergent entry) + +| host | trigger → calls `session_split.summarize` | +|---|---| +| claude code, codex | existing `vouch capture finalize` hook / ingest | +| openclaw | its `compact` rpc — the "history is huge" signal (fast-follow) | +| mcp-only hosts | new `kb.summarize_session` method | +| all hosts | `finalize-all` stale-buffer sweep on next start (already host-blind) | + +### config (neutral, deployment-level) + +```yaml +capture: + split: + enabled: true + llm_cmd: null # defaults to compile.llm_cmd when unset + threshold_observations: 40 + max_pages: 6 + timeout_seconds: 180 + max_input_chars: 60000 +``` + +## scope + +**v1 (this plan):** + +- `session_split.py` host-blind core with the three-tier gate and the llm + topical split +- `llm_draft.py` extracted from `compile.py`; `compile.py` refactored to import it +- `capture.finalize` delegating to the core; `--split/--no-split` cli flag +- `kb.summarize_session` across all four surfaces + parity test +- neutral intent resolution with claude code's transcript parser demoted to an + optional enricher +- config block + tests + +**fast-follow (separate plan):** + +- wiring openclaw's `compact` rpc to `session_split.summarize` +- map-reduce chunking for buffers that exceed `max_input_chars` + +**out of scope:** + +- splitting into durable topic pages (`concept`/`workflow`/`decision`) — that is + `compile.py`'s job, from cited claims +- an index/parent page linking the children (flat topical pages only) +- auto-approval of any produced page + +## testing — `tests/test_session_split.py` + +- size gate: `total < threshold` → one mechanical page, `mode="mechanical"` +- size gate: `total ≥ threshold` with a fake `llm_cmd` echoing canned json → N + `PENDING` proposals, all `type=session`, all pending, `mode="split"` +- fallback: threshold met but no `llm_cmd` → mechanical, `mode="fallback"` +- fallback: llm returns junk / nonzero exit / timeout → mechanical, `mode="fallback"` +- dedupe: a draft whose title collides with an existing page is dropped +- cap: more than `max_pages` drafts → capped, extras in `dropped` +- host-neutral: observations carrying non-claude-code tool names (e.g. + `fs.write`, `shell.exec`) cluster without crashing +- intent priority: `Session.task` used when present, filename fallback when absent +- back-compat: `finalize()` still returns a `summary_proposal_id` key +- `kb.summarize_session` behavior + `test_capabilities` parity + +fake-`llm_cmd` fixture mirrors the existing compile tests: a shell command that +echoes a canned json array on stdout. diff --git a/docs/superpowers/specs/2026-07-10-session-transcript-viewer-design.md b/docs/superpowers/specs/2026-07-10-session-transcript-viewer-design.md new file mode 100644 index 00000000..63d65bef --- /dev/null +++ b/docs/superpowers/specs/2026-07-10-session-transcript-viewer-design.md @@ -0,0 +1,352 @@ +# Session Transcript Viewer — Design + +Date: 2026-07-10 +Status: Implemented (see `docs/superpowers/plans/2026-07-10-session-transcript-viewer.md`) +Repo: `vouch` (backend `src/vouch`, frontend `webapp`) + +## Problem + +The vouch console (`webapp`) can list captured agent sessions +(`kb.list_sessions`) but cannot show what actually happened inside one. A row +gives a title, a stage, and an observation count — nothing more. When an agent +files claims a reviewer often needs to see the reasoning: the prompts, the +assistant's replies, the tools it ran, and the diffs it made. + +The `agentsview` project already renders exactly this for Claude Code / Codex +sessions. This feature ports **agentsview's rendering experience** into the +vouch console so a reviewer can open a session and read its full transcript. + +### What this is NOT + +- Not live-run rendering. The chat's "Claude mode" (`ChatView`) stays as-is; + this feature is a read-only viewer for **already-captured** sessions. +- Not a re-implementation of agentsview's storage architecture. See + "Relationship to agentsview" below — we copy the rendering, not the + sync-into-a-database pipeline. + +## Relationship to agentsview (what we copy, what we don't) + +agentsview works in two stages: + +1. **Ingest → SQLite.** A sync engine + file watcher parse the raw agent JSONL + into normalized rows in a SQLite DB (`messages`, `tool_calls`, + `tool_result_events`, FTS5). The raw file is parsed once at sync time. +2. **Serve → render.** A paginated REST API reads from the DB; the Svelte + frontend segments `content` + `tool_calls` into typed blocks + (`thinking` / `tool` / `code` / `skill` / `text`) client-side and renders + them with per-block components. + +We faithfully copy **stage 2's rendering vocabulary** (block types, per-tool +rendering, diffs, collapsibles, lazy subagents). We deliberately do **not** +copy stage 1: instead of syncing raw files into a database, the vouch backend +**parses the raw file on demand** when a session is opened. This yields the +same on-screen result with none of the sync/DB/watcher machinery, which is the +right trade for a viewer. + +Consequences accepted for v1: + +- Re-parses on each open (fine — a viewer opens one session at a time; large + sessions are handled by capping + lazy subagent loading, below). +- No cross-session full-text search inside transcripts. +- If the raw file has been deleted, the viewer degrades to vouch's compact + observations (below) rather than failing. + +## Scope + +In scope for v1: + +- Agents: **Claude Code** (`~/.claude/projects//.jsonl`) and + **Codex** (rollouts under `$CODEX_HOME/sessions/...`), on the **same machine** + as the vouch server. +- A new read-only RPC `kb.session_transcript`. +- Full-fidelity transcript rendering in the console's **Review** page + (master–detail): the session list plus the selected session's transcript and + its `Summarize` action, side by side. (Originally shipped as a standalone + **Sessions** tab, then merged into Review — see the update note below.) + +Out of scope for v1: + +- Remote / multi-machine sessions whose raw files are not on the server host + (would require a transcript-upload pipeline). +- Persisting or indexing transcripts; cross-session search. +- Editing, pinning, exporting, or analytics over transcripts. + +## Backend + +### New RPC: `kb.session_transcript` + +Params: + +- `session_id` (string, required) — the captured session id. +- `agent` (string, optional) — `"claude"` | `"codex"`. When omitted, the + locator auto-detects by searching both sources. + +Success result (raw file found and parsed): + +```jsonc +{ + "available": true, + "source": { "agent": "claude", "path": "/home/u/.claude/projects/.../.jsonl" }, + "session": { + "id": "…", + "cwd": "…", + "git_branch": "…", + "started_at": "ISO-8601", + "ended_at": "ISO-8601", + "model": "…", + "tokens": { "input": 0, "output": 0, "cache_read": 0, "cache_creation": 0 } + }, + "messages": [ + { + "role": "user" | "assistant", + "id": "…", // message id (assistant), else null + "model": "…", // optional, assistant only + "timestamp": "ISO-8601", // optional + "tokens": { "input": 0, "output": 0, "cache_read": 0, "cache_creation": 0 }, + "blocks": [ + { "type": "text", "text": "…" }, + { "type": "thinking", "text": "…" }, + // tool results are paired into their tool_use block server-side, so + // the frontend renders input + output together (agentsview's ToolBlock). + { "type": "tool_use", "id": "…", "name": "Bash", "input": { }, + "result": { "content": "…", "is_error": false, + "subagent_session_id": null } } // null until paired + ] + } + ], + "truncated": false // true if message cap hit (see limits) +} +``` + +Implementation note: the draft's separate `tool_result` block was dropped in +favor of pairing each `tool_result` into its originating `tool_use` block by id +during the single parse pass. The tool-result-only `user` entries are consumed, +not emitted as standalone messages. This matches agentsview's unified tool +block (input + output shown together) and keeps the frontend renderer simple. + +Degraded result (raw file unavailable — deleted, off-machine, unreadable): + +```jsonc +{ + "available": false, + "reason": "raw transcript not found for session ", + "observations": [ { "ts": "…", "tool": "Edit", "summary": "Edited types.go" } ] +} +``` + +`observations` are read from vouch's existing capture buffer +(`capture._read_observations`) when one still exists for the session; otherwise +an empty list. This is the honest fallback: it is agentsview's *rendering* +degraded to vouch's *data*. + +Errors (structured JSONL envelope, matching every other handler): + +- missing `session_id` → `missing_param` (a bare `p["session_id"]` KeyError, + mapped by the dispatcher). +- `agent` present but not `claude`/`codex` → `invalid_request` (a raised + `ValueError`). + +Registered on all three surfaces to satisfy vouch's parity invariant +(`test_capabilities`): `jsonl_server.HANDLERS`, `capabilities.METHODS`, and the +`kb_session_transcript` MCP tool in `server.py`. `http_server` reuses +`jsonl_server.handle_request`, so it is covered by the JSONL registration. + +Locator env overrides (used by tests, honored in prod): `VOUCH_CLAUDE_PROJECTS_DIR` +re-roots the Claude search; `CODEX_HOME` re-roots the Codex rollout search. + +### New module: `src/vouch/transcript.py` (pure, fixture-tested) + +Responsibilities, each a small pure function: + +1. **Locate** the raw file for `(session_id, agent?)`: + - Claude Code: glob `~/.claude/projects/*/.jsonl` (the file stem + is the session id). Subagent files live at + `~/.claude/projects/*//subagents/**/*.jsonl`. + - Codex: reuse `codex_rollout.find_rollout_by_session_id`. + - Auto-detect tries Claude then Codex. + - Honor `VOUCH_CLAUDE_PROJECTS_DIR` / `CODEX_HOME` overrides for tests. + - `session_id` is validated against a UUID-shaped pattern before any glob so + a hostile id cannot widen the search or traverse the tree. +2. **Parse + normalize** raw lines into the `messages[]` schema above. + - Claude Code line schema (per JSONL entry): `message.role`, + `message.model`, `message.usage.{input_tokens,output_tokens, + cache_read_input_tokens,cache_creation_input_tokens}`, and + `message.content[]` whose parts have `type` ∈ + `{text, thinking, tool_use, tool_result}`. `tool_use` carries + `{id, name, input}`; `tool_result` (in user entries) carries + `{tool_use_id, content, is_error}`. Session-level `cwd`, `gitBranch`, + `timestamp` come from the entries. Subagent linkage via + `toolUseResult.agentId` maps a `tool_result` to its child session id + (mirrors agentsview's `subagentMap`). + - Codex: `codex_rollout.parse_rollout` is lossy (compact observations only), + so a dedicated `parse_codex_transcript` reads the raw `response_item` + records — the canonical conversation stream: `message` (role user → + `input_text`, assistant → `output_text`; developer/system boilerplate + skipped), `function_call` / `custom_tool_call` + their `*_output` pairs, + and `reasoning` (encrypted, so dropped). `session_meta` supplies + `id`/`cwd`/`git.branch`/`timestamp`. `event_msg` records are UI mirrors and + ignored to avoid duplication. + - Malformed lines are skipped, not fatal (matches the existing stream + parser's tolerance). +3. **Limits** (protect the server + browser): + - Max file size read (config const, e.g. 25 MB); over → degraded result with + reason. + - Max messages returned (config const, e.g. 2000); over → `truncated: true`. + - Per-block content is passed through; very large tool outputs are the + browser's problem to collapse, not the server's to trim (agentsview keeps + full content; we match). + +Subagents are fetched lazily by a **second** `kb.session_transcript` call with +the child `session_id` (the frontend passes `subagent_session_id`). The backend +locator finds `~/.claude/projects/*//subagents/**/.jsonl` as well +as top-level files, so the same RPC serves both. + +### Wiring + +- Handler `_h_session_transcript(p)` registered in + `src/vouch/jsonl_server.py` `HANDLERS` (and the HTTP surface in + `http_server.py` if it maintains its own map). +- Add `"kb.session_transcript"` to `src/vouch/capabilities.py` method list + (the `test_capabilities` drift test enforces this). +- Read-only: never calls `approve`/`propose`; unaffected by the review gate. + +## Frontend (`webapp`, React + Tailwind v4) + +Port agentsview's block vocabulary into React, styled with vouch's existing +semantic tokens (`paper` / `ink` / `accent` / `sepia` / `rule` / `ok`), reusing +the existing `Markdown` component for text blocks and `lucide-react` icons. + +### Route + entry point + +Post-merge (see the update note below), the viewer lives in the existing +**Review** page rather than a separate tab: + +- No new route or nav item. `ReviewView` (`/review`) is the single sessions + surface. Left: all sessions from `kb.list_sessions` (`useFanout`) — the + earlier `!summarized` filter was dropped so every session's transcript stays + viewable. Right (detail pane): a compact header with the session's stage / + ids / `Summarize` action, and below it `TranscriptView` for the selected + session — gated on `hasMethod('kb.session_transcript')` and a non-null + `session_id`, degrading to a note otherwise. +- The `Summarize` action shows only for sessions that still need it + (`!summarized`); already-summarized rows render read-only. + +### Components (each maps to an agentsview equivalent) + +| vouch (new) | agentsview reference | behavior | +| --- | --- | --- | +| `TranscriptView` | `MessageList` | fetches `kb.session_transcript(id)`, renders `messages[]`; shows session vitals header (model, tokens, cwd, branch) | +| `MessageBlock` | `MessageContent` | role chrome, model badge, tokens, timestamp; dispatches blocks | +| `ThinkingBlock` | `ThinkingBlock` | collapsible "Thinking" | +| `ToolBlock` | `ToolBlock` | collapsible; per-tool rendering; error styling | +| `DiffView` | ToolBlock diff-view | +/− line rendering for Edit/Write | +| `CodeBlock` | `CodeBlock` | fenced code | +| `TextBlock` | markdown path | reuse existing `Markdown` | + +Per-tool rendering inside `ToolBlock` (parity with agentsview): + +- `Bash` / `run_command` → command line + collapsible stdout. +- `Edit` / `Update` / `MultiEdit` → `DiffView`. +- `Write` → created-file content. +- `Read` / `Grep` / `Glob` → compact summary (path/pattern) + collapsible body. +- `Task` / `Agent` → labeled subagent step; when the paired result carries a + `subagent_session_id`, a **"view subagent"** button pushes the child onto an + in-`TranscriptView` back-stack and re-fetches `kb.session_transcript(child)`. +- Unknown tools (and every tool's raw input) → collapsible pretty-printed JSON. + +Shipped simplification: `Read`/`Grep`/`Glob` render a one-line headline plus the +collapsible output; a dedicated `TodoWrite` checklist renderer was deferred +(TodoWrite falls through to the JSON input view). Easy follow-up if wanted. + +### Degraded rendering + +When `available === false`, `TranscriptView` shows a notice ("original +transcript unavailable — showing captured activity") and renders the +`observations` as a compact tool timeline. + +### Client library + +- `webapp/src/lib/transcript.ts` — types for the normalized schema + a thin + `fetchTranscript(conn, id)` wrapper over `rpc('kb.session_transcript', …)`. + No new transport; reuses `/proxy/rpc`. + +## Error handling summary + +- Unknown / null session id → structured RPC error surfaced as a `Toast` + an + `ErrorCard` in the detail pane. +- File missing/oversized/unreadable → `available:false` degraded result. +- Malformed transcript lines → skipped in the parser. +- Endpoint doesn't advertise the method → nav hidden / disabled (capability + gate), never a hard failure. + +## Testing + +Backend (vouch conventions: `pytest`, `mypy src`, `ruff check`): + +- `tests/test_session_transcript.py` — table/fixture tests over the parser with + small committed **Claude Code** and **Codex** JSONL fixtures: text, thinking, + tool_use→tool_result pairing, is_error, subagent linkage, model/tokens, cwd/ + branch extraction; malformed-line tolerance; size/message caps → `truncated`. +- Locator tests using `VOUCH_CLAUDE_PROJECTS_DIR` / `CODEX_HOME` pointed at + `tmp_path` fixtures; auto-detect order; subagent-file resolution. +- Degradation test: no raw file, buffer present → `available:false` + + observations; no raw file, no buffer → empty observations. +- RPC envelope test `tests/test_session_transcript.py` asserting the JSONL + envelope shape; capabilities drift covered by `test_capabilities`. + +Frontend (`vitest` + Testing Library; one Playwright smoke): + +- Component tests per block (`ToolBlock` per-tool branches incl. `DiffView`, + `ThinkingBlock` collapse, `Task` subagent lazy-load with a mocked rpc), + `TranscriptView` happy path + degraded path + subagent drill-down, + `ReviewView` (all sessions listed incl. summarized, transcript in the detail + pane, `Summarize` still works alongside it). +- `webapp/e2e/` smoke: open Sessions, pick a row, see the rendered transcript. + It stubs `/proxy/*` via Playwright `page.route` (health, capabilities, + `kb.list_pending`, `kb.list_sessions`, `kb.session_transcript`) so it drives + the real frontend independent of the backend build — the local `vouch` on + PATH is an editable install of a different checkout without the new RPC. +- Tests assert rendered behavior, not implementation strings + (`testing-without-tautologies`). + +## Conventions / guardrails (this repo) + +- Follow `vouch` `AGENTS.md`, not agentsview's `CLAUDE.md`: conventional + commits `(): …`; run `pytest --ignore=tests/embeddings`, + `mypy src`, `ruff check src tests` before shipping. **No + `Co-Authored-By: ` trailer.** No secrets/paths-as-PII in commits. +- Work on a feature branch (ask before creating it); do not commit to `main` + without permission; do not merge. + +## Phasing + +1. Backend: `transcript.py` (Claude locator + parser) + RPC + capabilities + + tests. +2. Frontend: `TranscriptView` + block components + client lib + tests, wired to + phase-1 RPC. (Initially surfaced via a `SessionsView` tab.) +3. Codex source (reuse `codex_rollout`) + subagent lazy expansion + degraded + fallback + e2e smoke. +4. Merge: fold the transcript into `ReviewView`, drop the `!summarized` filter, + and remove the standalone Sessions tab (this iteration). + +## Resolved (as shipped) + +- Entry point: **merged into the Review page** (`/review`). The initial cut + shipped a standalone **Sessions** tab, but Review already listed the same + `kb.list_sessions` sessions (to summarize them) with no transcript, so the two + were redundant. Review now renders the transcript in its detail pane next to + the `Summarize` action; the Sessions tab/route/view were removed. +- List scope: Review shows **all** sessions (its `!summarized` filter was + dropped), so transcripts of already-summarized sessions stay viewable; the + `Summarize` action is contextual (only for sessions that still need it). +- Layout: master–detail — session list on the left, transcript + summarize on + the right. +- Codex shipped in v1 (phase 3) alongside Claude Code, via a dedicated + `response_item` parser (not the lossy `parse_rollout`). + +## Deferred (possible follow-ups) + +- Dedicated `TodoWrite` checklist renderer (currently the JSON fallback). +- Rendering `reasoning`/`thinking` when a session persists plaintext (VSCode/SDK + sessions store only the encrypted signature, so thinking blocks are dropped). +- Remote / multi-machine transcript access (still out of scope). diff --git a/hatch_build.py b/hatch_build.py new file mode 100644 index 00000000..c47df8d6 --- /dev/null +++ b/hatch_build.py @@ -0,0 +1,27 @@ +"""Bundle the built React console (webapp/dist) into the wheel. + +When webapp/dist has been built, ship it as ``vouch/web/console`` so +``pip install 'vouch-kb[web]'`` + ``vouch console`` serves it with no node. +When it has NOT been built (a fresh checkout, or the sdist -> wheel path where +the gitignored dist isn't present), skip it silently: the wheel still builds, +and ``vouch console`` reports the missing console cleanly. This is why the +include is a hook rather than a static ``force-include``, which hard-errors on +a missing source path. +""" + +from __future__ import annotations + +import os +from typing import Any + +from hatchling.builders.hooks.plugin.interface import BuildHookInterface + + +class ConsoleBundleHook(BuildHookInterface): + PLUGIN_NAME = "custom" + + def initialize(self, version: str, build_data: dict[str, Any]) -> None: + dist = os.path.join(self.root, "webapp", "dist") + if not os.path.isfile(os.path.join(dist, "index.html")): + return # not built — degrade gracefully rather than fail the build + build_data.setdefault("force_include", {})[dist] = "vouch/web/console" diff --git a/llms.txt b/llms.txt index 3d3e9c3e..66482ee0 100644 --- a/llms.txt +++ b/llms.txt @@ -21,7 +21,8 @@ Repo: https://github.com/vouchdev/vouch · PyPI: `vouch-kb` · CLI: `vouch` ## Quick start and onboarding -- [docs/getting-started.md](https://raw.githubusercontent.com/vouchdev/vouch/main/docs/getting-started.md): Agent-side loop — register a source, propose a claim, get it approved. +- [docs/INSTALL_FOR_AGENTS.md](https://raw.githubusercontent.com/vouchdev/vouch/main/docs/INSTALL_FOR_AGENTS.md): Imperative, agent-followable install checklist — detect host, `install-mcp`, verify `capabilities`, init KB, propose → (human) approve smoke test. Distinct from the human-facing getting-started walkthrough below. +- [docs/getting-started.md](https://raw.githubusercontent.com/vouchdev/vouch/main/docs/getting-started.md): Human-oriented quickstart — register a source, propose a claim, get it approved. - [docs/example-session.md](https://raw.githubusercontent.com/vouchdev/vouch/main/docs/example-session.md): End-to-end captured walkthrough that the demo gif is recorded from. - [docs/faq.md](https://raw.githubusercontent.com/vouchdev/vouch/main/docs/faq.md): "Why not Notion / Obsidian / a vector DB" and other recurring asks. diff --git a/openclaw.plugin.json b/openclaw.plugin.json index ffcd470d..a21192b1 100644 --- a/openclaw.plugin.json +++ b/openclaw.plugin.json @@ -1,7 +1,7 @@ { "id": "vouch", "name": "Vouch", - "version": "1.2.2", + "version": "1.3.0", "description": "Git-native, review-gated knowledge base. Registers vouch's context engine (cited retrieval + salience reflex + hot memory) and the vouch skills. The kb.* MCP server is deployment config: `openclaw mcp add vouch -- vouch serve`.", "kind": "context-engine", "skills": [ diff --git a/package.json b/package.json index 9ccee39b..f62c5cb0 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "vouch", - "version": "1.2.2", + "version": "1.3.0", "private": true, "description": "OpenClaw plugin packaging for vouch. The Python package lives in pyproject.toml; this file only tells OpenClaw's plugin loader which entry module to import and which plugin API range the plugin supports.", "openclaw": { diff --git a/pyproject.toml b/pyproject.toml index 624cf336..c15a6260 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "vouch-kb" -version = "1.2.2" +version = "1.3.0" description = "Git-native, review-gated knowledge base for LLM agents. MCP server + CLI." readme = "README.md" requires-python = ">=3.11" @@ -82,6 +82,14 @@ packages = ["src/vouch"] [tool.hatch.build.targets.wheel.force-include] "adapters" = "vouch/adapters" +# The built React console (webapp/dist) rides inside the wheel as +# vouch/web/console so `pip install 'vouch-kb[web]'` + `vouch console` serves it +# with no node. It is force-included *conditionally* by hatch_build.py — a plain +# force-include hard-errors when webapp/dist isn't built (fresh checkout, sdist +# → wheel), whereas the hook skips it and `vouch console` reports it cleanly. +[tool.hatch.build.targets.wheel.hooks.custom] +path = "hatch_build.py" + [tool.ruff] line-length = 100 target-version = "py311" diff --git a/schemas/capabilities.schema.json b/schemas/capabilities.schema.json index f795ed3d..118969f0 100644 --- a/schemas/capabilities.schema.json +++ b/schemas/capabilities.schema.json @@ -29,6 +29,12 @@ "title": "Level", "type": "integer" }, + "mcp": { + "additionalProperties": true, + "description": "mcp surface flags mirrored from config.yaml `mcp:` block. `publish_skills` gates kb.list_skills / kb.get_skill.", + "title": "Mcp", + "type": "object" + }, "methods": { "items": { "type": "string" diff --git a/schemas/claim.schema.json b/schemas/claim.schema.json index 2e3ac274..84cdd764 100644 --- a/schemas/claim.schema.json +++ b/schemas/claim.schema.json @@ -89,6 +89,11 @@ "default": null, "title": "Approved By" }, + "auto_approved": { + "default": false, + "title": "Auto Approved", + "type": "boolean" + }, "confidence": { "default": 0.7, "maximum": 1.0, @@ -141,6 +146,18 @@ "default": null, "title": "Last Confirmed At" }, + "proposed_by": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "title": "Proposed By" + }, "scope": { "$ref": "#/$defs/ArtifactScope" }, diff --git a/schemas/proposal.schema.json b/schemas/proposal.schema.json index 26c1157b..5fcbf5dc 100644 --- a/schemas/proposal.schema.json +++ b/schemas/proposal.schema.json @@ -5,7 +5,8 @@ "claim", "page", "entity", - "relation" + "relation", + "delete" ], "title": "ProposalKind", "type": "string" diff --git a/scripts/console.sh b/scripts/console.sh new file mode 100755 index 00000000..f6c9f8b2 --- /dev/null +++ b/scripts/console.sh @@ -0,0 +1,84 @@ +#!/usr/bin/env bash +# +# Run the vouch backend and the vouch-ui web console together. +# +# Starts `vouch serve --transport http` (127.0.0.1:8731) and the vouch-ui Vite +# dev server (webapp/, http://localhost:5173) as a pair, and shuts both down on +# Ctrl-C. The dev server proxies the UI's /proxy/* calls to the vouch endpoint +# (via the X-Vouch-Target header), so the browser talks only to the dev server +# and vouch itself stays unmodified and needs no CORS. +# +# Usage: +# make console # from the repo root +# scripts/console.sh # equivalent +# VOUCH=vouch scripts/console.sh # use a `vouch` on PATH +# VOUCH_HOST=0.0.0.0 VOUCH_PORT=8731 scripts/console.sh +# +# Requires: node + npm (for the UI). Dependencies install automatically on the +# first run. Open http://localhost:5173 and connect to http://127.0.0.1:8731. + +set -uo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +WEBAPP="$ROOT/webapp" +HOST="${VOUCH_HOST:-127.0.0.1}" +PORT="${VOUCH_PORT:-8731}" + +# Chat's "Claude Code mode" spawns `claude -p` in a workspace; default it to the +# repo root (the vendored bridge's own default of ../vouch is wrong from webapp/). +export VOUCH_PROJECT_DIR="${VOUCH_PROJECT_DIR:-$ROOT}" + +# Prefer the repo virtualenv's vouch, then one on PATH. Override with $VOUCH. +if [ -z "${VOUCH:-}" ]; then + if [ -x "$ROOT/.venv/bin/vouch" ]; then + VOUCH="$ROOT/.venv/bin/vouch" + else + VOUCH="vouch" + fi +fi + +if [ ! -d "$WEBAPP" ]; then + echo "console: $WEBAPP not found — the webapp/ folder is missing" >&2 + exit 1 +fi +if ! command -v npm >/dev/null 2>&1; then + echo "console: npm not found — install Node.js to run the web console" >&2 + exit 1 +fi + +# First run: install the web console's dependencies. +if [ ! -d "$WEBAPP/node_modules" ]; then + echo "[console] installing web console dependencies (first run, may take a minute)…" + (cd "$WEBAPP" && npm install) || { echo "console: npm install failed" >&2; exit 1; } +fi + +pids=() +cleanup() { + trap - EXIT INT TERM HUP + echo + echo "[console] shutting down…" + for pid in "${pids[@]}"; do + # Each service is started with `setsid`, so its pid is a process-group + # leader: kill the whole group (negative pid) to take down grandchildren + # too (e.g. npm -> vite), then fall back to the bare pid. + kill -TERM "-$pid" 2>/dev/null || kill -TERM "$pid" 2>/dev/null || true + done + wait 2>/dev/null || true +} +trap cleanup EXIT INT TERM HUP + +echo "[console] backend : $VOUCH serve --transport http -> http://$HOST:$PORT" +# shellcheck disable=SC2086 +setsid $VOUCH serve --transport http --host "$HOST" --port "$PORT" & +pids+=($!) + +echo "[console] frontend: vouch-ui dev server -> http://localhost:5173" +setsid bash -c 'cd "$1" && exec npm run dev' _ "$WEBAPP" & +pids+=($!) + +echo +echo "[console] both running. Open http://localhost:5173 (endpoint http://$HOST:$PORT)" +echo "[console] press Ctrl-C to stop both." + +# Return (and tear both down via the trap) as soon as either process exits. +wait -n 2>/dev/null || wait diff --git a/scripts/repro_list_pages_crash.py b/scripts/repro_list_pages_crash.py new file mode 100644 index 00000000..6cc35944 --- /dev/null +++ b/scripts/repro_list_pages_crash.py @@ -0,0 +1,78 @@ +#!/usr/bin/env python3 +"""Proof for PR #360: one corrupt pages/*.md crashes bulk listing on main. + +Run from the repo root:: + + python scripts/repro_list_pages_crash.py + +On **main** (before the fix), ``store.list_pages()`` raises:: + + ValueError: page file missing YAML frontmatter + +That exception propagates through ``health.lint()``, ``health.status()``, +``kb.list_pages``, and any other caller of ``list_pages()`` — one bad file +takes down the whole KB listing surface. + +With the fix branch applied, ``list_pages()`` logs a warning and returns +only the readable pages. +""" + +from __future__ import annotations + +import shutil +import sys +import tempfile +from pathlib import Path + +from vouch import health +from vouch.models import Claim, Page, PageType +from vouch.storage import KBStore, _deserialize_page + + +def main() -> int: + root = Path(tempfile.mkdtemp(prefix="vouch-repro-list-pages-")) + try: + store = KBStore.init(root) + src = store.put_source(b"evidence") + store.put_claim(Claim(id="c1", text="fact", evidence=[src.id])) + store.put_page(Page(id="good-page", title="Good", body="ok", type=PageType.CONCEPT)) + bad_path = store.kb_dir / "pages" / "bad-page.md" + bad_path.write_text("not valid frontmatter — no YAML block", encoding="utf-8") + + print("=== 1. Direct deserialize (always fails on corrupt file) ===") + try: + _deserialize_page(bad_path.read_text(encoding="utf-8")) + print("unexpected: _deserialize_page succeeded") + except ValueError as e: + print(f"ValueError: {e}") + + print("\n=== 2. store.list_pages() ===") + try: + pages = store.list_pages() + print(f"OK — returned {len(pages)} page(s): {[p.id for p in pages]}") + print("(fix branch: bad-page.md skipped with a logged warning)") + except ValueError as e: + print(f"CRASH — ValueError: {e}") + print("(main branch: entire listing aborts here)") + + print("\n=== 3. health.lint(store) ===") + try: + report = health.lint(store) + print(f"OK — lint finished, ok={report.ok}, findings={len(report.findings)}") + except ValueError as e: + print(f"CRASH — ValueError: {e}") + + print("\n=== 4. health.status(store) ===") + try: + summary = health.status(store) + print(f"OK — pages={summary['pages']}") + except ValueError as e: + print(f"CRASH — ValueError: {e}") + + return 0 + finally: + shutil.rmtree(root, ignore_errors=True) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/vouch/__init__.py b/src/vouch/__init__.py index 236c24ff..c0b9743c 100644 --- a/src/vouch/__init__.py +++ b/src/vouch/__init__.py @@ -3,4 +3,4 @@ # One of FOUR version sites kept in lockstep (with pyproject.toml, # openclaw.plugin.json, package.json) — enforced by # tests/test_openclaw_plugin_manifest.py::test_manifest_versions_in_step. -__version__ = "1.2.2" +__version__ = "1.3.0" diff --git a/src/vouch/audit.py b/src/vouch/audit.py index e39703ba..a37bb1ca 100644 --- a/src/vouch/audit.py +++ b/src/vouch/audit.py @@ -7,6 +7,7 @@ from __future__ import annotations +import contextlib import hashlib import json import os @@ -23,6 +24,7 @@ from .storage import KBStore AUDIT_FILENAME = "audit.log.jsonl" +AUDIT_LOCKFILE = AUDIT_FILENAME + ".lock" GENESIS_HASH = "0" * 64 @@ -30,6 +32,45 @@ def _audit_path(kb_dir: Path) -> Path: return kb_dir / AUDIT_FILENAME +def _audit_lockfile(kb_dir: Path) -> Path: + return kb_dir / AUDIT_LOCKFILE + + +@contextlib.contextmanager +def _audit_lock(kb_dir: Path) -> Iterator[None]: + """Hold an exclusive cross-process lock for the duration of a log_event. + + Serialises the read-then-append sequence in `log_event` so two concurrent + writers cannot both observe the same `prev_hash` and fork the chain. + Lock is held on a sibling `audit.log.jsonl.lock` file so the audit log + itself is never opened in a mode that could truncate it. Blocks until + acquired and is always released, including on exceptions. + """ + lockfile = _audit_lockfile(kb_dir) + lockfile.parent.mkdir(parents=True, exist_ok=True) + fd = os.open(lockfile, os.O_RDWR | os.O_CREAT, 0o644) + try: + if os.name == "posix": + fcntl = __import__("fcntl") + fcntl.flock(fd, fcntl.LOCK_EX) + try: + yield + finally: + fcntl.flock(fd, fcntl.LOCK_UN) + else: + msvcrt = __import__("msvcrt") + os.lseek(fd, 0, os.SEEK_SET) + msvcrt.locking(fd, msvcrt.LK_LOCK, 1) + try: + yield + finally: + os.lseek(fd, 0, os.SEEK_SET) + with contextlib.suppress(OSError): + msvcrt.locking(fd, msvcrt.LK_UNLCK, 1) + finally: + os.close(fd) + + def new_event_id() -> str: return uuid.uuid4().hex @@ -75,28 +116,35 @@ def log_event( reversible: bool = True, data: dict[str, Any] | None = None, ) -> AuditEvent: - """Append one AuditEvent. Returns the persisted event.""" - prev_hash = _last_hash(kb_dir) - ev = AuditEvent( - id=new_event_id(), - event=event, - actor=actor, - object_ids=object_ids or [], - dry_run=dry_run, - reversible=reversible, - data=data or {}, - prev_hash=prev_hash, - ) - ev.hash = _compute_hash(prev_hash, _event_payload_for_hash(ev)) + """Append one AuditEvent. Returns the persisted event. + + Holds an exclusive cross-process lock around read-prev-hash → derive → + append so concurrent writers can't fork the chain. Without the lock, two + log_event calls racing on the same KB observe the same prev_hash and both + chain off it — verify_chain then reports "previous hash mismatch" forever. + """ path = _audit_path(kb_dir) path.parent.mkdir(parents=True, exist_ok=True) - line = _canonical_json(ev.model_dump(mode="json")) - # Open-write-close for crash safety — if the process dies mid-append the - # log is still parseable up to the last newline. - with path.open("a", encoding="utf-8") as f: - f.write(line + "\n") - f.flush() - os.fsync(f.fileno()) + with _audit_lock(kb_dir): + prev_hash = _last_hash(kb_dir) + ev = AuditEvent( + id=new_event_id(), + event=event, + actor=actor, + object_ids=object_ids or [], + dry_run=dry_run, + reversible=reversible, + data=data or {}, + prev_hash=prev_hash, + ) + ev.hash = _compute_hash(prev_hash, _event_payload_for_hash(ev)) + line = _canonical_json(ev.model_dump(mode="json")) + # Open-write-close for crash safety — if the process dies mid-append + # the log is still parseable up to the last newline. + with path.open("a", encoding="utf-8") as f: + f.write(line + "\n") + f.flush() + os.fsync(f.fileno()) return ev diff --git a/src/vouch/capabilities.py b/src/vouch/capabilities.py index e051115d..bbf2f940 100644 --- a/src/vouch/capabilities.py +++ b/src/vouch/capabilities.py @@ -31,9 +31,11 @@ "kb.capabilities", "kb.status", "kb.stats", + "kb.activity", "kb.digest", "kb.search", "kb.neighbors", + "kb.experts", "kb.context", "kb.synthesize", "kb.read_page", @@ -54,6 +56,7 @@ "kb.propose_page", "kb.propose_entity", "kb.propose_relation", + "kb.propose_delete", "kb.approve", "kb.reject", "kb.reject_extracted", @@ -62,12 +65,16 @@ "kb.contradict", "kb.archive", "kb.confirm", + "kb.clear_claims", "kb.cite", "kb.source_verify", "kb.session_start", "kb.session_end", + "kb.list_sessions", + "kb.session_transcript", "kb.volunteer_context", "kb.crystallize", + "kb.summarize_session", "kb.index_rebuild", "kb.lint", "kb.doctor", @@ -88,6 +95,8 @@ "kb.detect_themes", "kb.propose_theme", "kb.compile", + "kb.list_skills", + "kb.get_skill", ] @@ -112,7 +121,7 @@ def _load_host_compat() -> dict[str, dict[str, str]]: return {"openclaw": {k: str(v) for k, v in compat.items()}} -def capabilities() -> Capabilities: +def capabilities(*, publish_skills: bool = True) -> Capabilities: retrieval = ["fts5", "substring"] try: from .embeddings import get_embedder @@ -134,5 +143,6 @@ def capabilities() -> Capabilities: "config_path": "retrieval.scope", }, context_engines=[describe_engine()], + mcp={"publish_skills": publish_skills}, host_compat=_load_host_compat(), ) diff --git a/src/vouch/capture.py b/src/vouch/capture.py index f87f01f9..06b56e38 100644 --- a/src/vouch/capture.py +++ b/src/vouch/capture.py @@ -21,7 +21,6 @@ import yaml from .models import ProposalStatus -from .proposals import propose_page from .storage import KBStore DEFAULT_ENABLED = True @@ -320,53 +319,26 @@ def finalize( project: str | None = None, generated_at: str | None = None, transcript_path: Path | None = None, + mode: str = "auto", config: CaptureConfig | None = None, ) -> dict[str, Any]: - """Roll a session buffer into one PENDING summary proposal. No approve(). - - If cwd is None (e.g., when finalizing orphaned buffers with unknown origin), - git changes are not included. Otherwise, git changes from cwd are included. - transcript_path (from the SessionEnd hook payload) supplies the human's - first prompt for the proposal title; absent, the title falls back to the - files the session touched. + """Roll a session buffer into PENDING summary proposal(s). No approve(). + + Claude-Code-facing wrapper: resolves the transcript's first user prompt + (its one host-specific enricher) and delegates to the host-blind + `session_split.summarize`. `mode` forwards "auto" | "split" | "mechanical". + If cwd is None (e.g., finalizing orphaned buffers of unknown origin), git + changes are not included; transcript_path (from the SessionEnd hook payload) + supplies the human's first prompt for the summary title when present. """ - cfg = config or load_config(store) - path = buffer_path(store, session_id) - observations = _read_observations(path) - if not cfg.enabled: - return {"captured": len(observations), "summary_proposal_id": None, - "skipped": "disabled"} - # Only include git context if cwd is explicitly provided (known origin) - # For cleanup of orphaned buffers, cwd=None, so skip git context - if cwd is not None: - changed_files, git_stat = _git_changes(cwd) - else: - changed_files, git_stat = [], "" - total = len(observations) + len(changed_files) - if total < cfg.min_observations: - if path.exists(): - path.unlink() - return {"captured": total, "summary_proposal_id": None, - "skipped": "below-min"} - first_prompt = ( + from . import session_split # deferred: breaks the capture<->session_split cycle + intent = ( first_user_prompt(transcript_path) if transcript_path is not None else None ) - title, body = build_summary_body( - session_id, observations, changed_files, git_stat, - project=project, generated_at=generated_at, first_prompt=first_prompt, - ) - proposal = propose_page( - store, - title=title, - body=body, - page_type=CAPTURE_PAGE_TYPE, - proposed_by=CAPTURE_ACTOR, - session_id=session_id, - rationale="auto-captured session summary", + return session_split.summarize( + store, session_id, intent=intent, cwd=cwd, project=project, + generated_at=generated_at, mode=mode, config=config, ) - if path.exists(): - path.unlink() - return {"captured": total, "summary_proposal_id": proposal.id} def pending_count(store: KBStore) -> int: diff --git a/src/vouch/cli.py b/src/vouch/cli.py index c6b04ccf..abe36889 100644 --- a/src/vouch/cli.py +++ b/src/vouch/cli.py @@ -11,6 +11,7 @@ import io import json import os +import sqlite3 import sys from collections.abc import Iterator from contextlib import contextmanager, suppress @@ -39,6 +40,7 @@ from . import provenance as prov_mod from . import recall as recall_mod from . import sessions as sess_mod +from . import skills as skills_mod from . import stats as stats_mod from . import sync as sync_mod from . import synthesize as synth @@ -236,7 +238,48 @@ def discover(path: str) -> None: @cli.command() def capabilities() -> None: """Emit the JSON capabilities descriptor (mirrors kb.capabilities).""" - _emit_json(build_caps().model_dump(mode="json")) + # Stay usable outside a KB: fall back to the default-on flag if no + # .vouch/ is discoverable here. + try: + publish_skills = skills_mod.publish_skills_enabled(_load_store()) + except Exception: + publish_skills = True + _emit_json(build_caps(publish_skills=publish_skills).model_dump(mode="json")) + + +@cli.command("list-skills") +@click.option("--json", "as_json", is_flag=True, help="Emit JSON instead of a table.") +def list_skills(as_json: bool) -> None: + """List discoverable Claude Code skills / slash commands (mirrors kb.list_skills).""" + store = _load_store() + rows = skills_mod.list_skills(store) + if as_json: + _emit_json(rows) + return + if not rows: + # Either nothing installed, or mcp.publish_skills is false. + click.echo("no skills published") + return + for r in rows: + click.echo(f"{r['name']} [{r['scope']}/{r['kind']}] {r['description']}") + + +@cli.command("get-skill") +@click.argument("name") +@click.option("--json", "as_json", is_flag=True, help="Emit JSON instead of the body.") +def get_skill(name: str, as_json: bool) -> None: + """Print the full body of a named skill / slash command (mirrors kb.get_skill).""" + store = _load_store() + try: + result = skills_mod.get_skill(store, name) + except skills_mod.SkillsDisabledError as e: + raise click.ClickException(str(e)) from e + except KeyError as e: + raise click.ClickException(str(e)) from e + if as_json: + _emit_json(result) + return + click.echo(result["body"]) # --- status / health ------------------------------------------------------ @@ -318,6 +361,59 @@ def stats(days: int, as_json: bool) -> None: _echo(f" invalid: {cites['invalid_claim']}, broken: {cites['broken_citation']}") +@cli.command() +@click.option( + "--days", + default=365, + show_default=True, + type=click.IntRange(min=0), + help="Window (local calendar days). Use 0 for all-time.", +) +@click.option( + "--tz-offset-minutes", + default=0, + show_default=True, + type=int, + help="Viewer's UTC offset in minutes for local-time bucketing.", +) +@click.option("--tz", default=None, help="IANA zone for local-time bucketing (wins over offset).") +@click.option("--project", default=None, help="Viewer project for audit scope filtering.") +@click.option("--agent", default=None, help="Viewer agent for audit scope filtering.") +@click.option("--json", "as_json", is_flag=True, help="Emit JSON instead of a table.") +def activity( + days: int, + tz_offset_minutes: int, + tz: str | None, + project: str | None, + agent: str | None, + as_json: bool, +) -> None: + """Audit activity buckets: per-day counts, hour-of-week matrix, actors.""" + from .scoping import viewer_from + + store = _load_store() + viewer = viewer_from( + config_path=store.config_path, + project=project, + agent=agent, + ) + body = stats_mod.collect_activity( + store, days=days, tz_offset_minutes=tz_offset_minutes, tz=tz, viewer=viewer, + ) + if as_json: + _emit_json(body) + return + window = "all time" if body["window_days"] is None else f"last {body['window_days']}d" + _echo( + f"activity ({window}): {_style(str(body['total_events']), fg='cyan')} events " + f"on {body['active_days']} day(s)" + ) + if body["first_event_day"]: + _echo(f" span: {body['first_event_day']} → {body['last_event_day']}") + for actor, count in list(body["by_actor"].items())[:8]: + _echo(f" {actor}: {count}") + + @cli.command(name="digest") @click.option( "--since", @@ -1571,6 +1667,38 @@ def new_cmd( return click.echo(pr.id) +@cli.command(name="experts") +@click.argument("topic") +@click.option("--limit", default=10, show_default=True, type=int) +@click.option("--min-claims", "min_claims", default=1, show_default=True, type=int) +@click.option( + "--weight", + default="count", + show_default=True, + help="ranking weight: count | recency | citation (unknown falls back to count).", +) +@click.option("--json", "as_json", is_flag=True, help="emit the ranking as JSON.") +def experts_cmd( + topic: str, limit: int, min_claims: int, weight: str, as_json: bool +) -> None: + """Rank entities by evidence density on TOPIC (read-only).""" + from .experts import rank_experts + + store = _load_store() + rows = rank_experts(store, topic, limit=limit, min_claims=min_claims, weight=weight) + if as_json: + _emit_json({"experts": rows}) + return + if not rows: + click.echo("no experts found.") + return + for row in rows: + click.echo( + f"{row['name']} ({row['type']}) " + f"claims={row['claim_count']} citations={row['citation_count']} " + f"score={row['score']}" + ) + @cli.group(name="schema") def schema() -> None: @@ -1889,6 +2017,69 @@ def archive(claim_id: str) -> None: click.echo(f"archived {claim_id}") +@cli.command(name="claims-clear") +@click.option("--auto-only", is_flag=True, default=True, show_default=True, + help="Clear only auto-approved claims (default: yes)") +@click.option("--before", type=str, default=None, + help="Clear only claims created before this date (ISO 8601, e.g. 2026-07-01)") +@click.option("--confirm", is_flag=True, default=False, + help="Skip confirmation prompt") +@click.option("--dry-run", is_flag=True, default=False, + help="Preview what would be cleared without making changes") +def claims_clear(auto_only: bool, before: str | None, confirm: bool, dry_run: bool) -> None: + """Clear auto-saved claims. Archived claims are preserved in history.""" + from datetime import datetime + + store = _load_store() + before_dt = None + if before: + try: + before_dt = datetime.fromisoformat(before) + except ValueError as err: + raise click.ClickException( + f"invalid date format: {before} (use ISO 8601, e.g. 2026-07-01)" + ) from err + + with _cli_errors(): + to_clear = life.clear_claims( + store, + auto_only=auto_only, + before=before_dt, + actor=_whoami(), + dry_run=True, # Always dry-run first to show what will be cleared + ) + + if not to_clear: + click.echo("no claims match the criteria") + return + + click.echo(f"found {len(to_clear)} claims to clear:") + for claim in to_clear[:10]: # Show first 10 + click.echo(f" {claim.id}: {claim.text[:60]}") + if len(to_clear) > 10: + click.echo(f" ... and {len(to_clear) - 10} more") + + if dry_run: + click.echo("(dry-run mode: no changes made)") + return + + if not confirm and not click.confirm(f"\nClear {len(to_clear)} claims?"): + click.echo("cancelled") + return + + # Now actually clear them + with _cli_errors(): + life.clear_claims( + store, + auto_only=auto_only, + before=before_dt, + actor=_whoami(), + dry_run=False, + ) + + click.echo(f"cleared {len(to_clear)} claims") + + @cli.command() @click.argument("claim_id") def confirm(claim_id: str) -> None: @@ -2312,14 +2503,20 @@ def search( ) used = "embedding" if hits else used if not hits and backend in ("auto", "fts5"): - hits = index_db.search(store.kb_dir, q, limit=fetch_limit) - used = "fts5" if hits else used + try: + hits = index_db.search(store.kb_dir, q, limit=fetch_limit) + used = "fts5" if hits else used + except sqlite3.Error: + hits = [] if not hits and backend in ("auto", "substring"): hits = store.search_substring(q, limit=fetch_limit) used = "substring" if backend == "hybrid": emb = index_db.search_semantic(store.kb_dir, q, limit=fetch_limit * 2) - fts = index_db.search(store.kb_dir, q, limit=fetch_limit * 2) + try: + fts = index_db.search(store.kb_dir, q, limit=fetch_limit * 2) + except sqlite3.Error: + fts = [] hits = rrf_fuse(emb, fts, limit=fetch_limit) used = "hybrid" @@ -2420,9 +2617,10 @@ def context( def context_hook() -> None: """Emit relevant KB context for a host UserPromptSubmit hook (reads stdin). - Wired by the claude-code adapter; not meant to be run by hand. Reads the - host's JSON hook payload on stdin, prints an additionalContext envelope, - and always exits 0 so it can never block a turn. + Wired by the claude-code and codex adapters (#425); not meant to be run + by hand. Both hosts emit the same {"prompt", "session_id", ...} shape on + stdin and expect the same additionalContext envelope back, so one + command serves both. Always exits 0 so it can never block a turn. """ import sys @@ -2444,12 +2642,17 @@ def context_hook() -> None: @click.argument("query") @click.option("--depth", default=3, show_default=True, type=int) @click.option("--max-chars", default=4000, show_default=True, type=int) -def synthesize(query: str, depth: int, max_chars: int) -> None: - """Answer a query from approved claims only, with inline citations.""" +@click.option( + "--llm", "use_llm", is_flag=True, + help="Draft the answer with the configured compile.llm_cmd, grounded in " + "pages and approved claims (citations still verified mechanically).", +) +def synthesize(query: str, depth: int, max_chars: int, use_llm: bool) -> None: + """Answer a query from the KB, with inline citations.""" store = _load_store() with _cli_errors(): result = synth.synthesize( - store, query=query, depth=depth, max_chars=max_chars, + store, query=query, depth=depth, max_chars=max_chars, llm=use_llm, ) _emit_json(result) @@ -3551,12 +3754,21 @@ def install_mcp( click.echo(f" ~ {f} (merged into existing)") for f in result.skipped: click.echo(f" · {f} (already present)") + for f in result.failed: + click.echo(f" ✗ {f} (could not install — left unchanged)") click.echo( f"Done — {len(result.written)} written, " f"{len(result.appended)} appended, {len(result.merged)} merged, " - f"{len(result.skipped)} skipped " + f"{len(result.skipped)} skipped, {len(result.failed)} failed " f"under {target}" ) + if result.failed: + # a failed install is not a no-op: exit non-zero so scripts (and the + # user) notice vouch was NOT wired into these files. + raise click.ClickException( + f"{len(result.failed)} file(s) could not be installed: " + f"{', '.join(result.failed)}" + ) # --- sync: bidirectional vouch <-> Obsidian-style vault ------------------- @@ -3688,6 +3900,71 @@ def _resolve_auth_token(auth: str | None) -> str | None: return auth +@cli.command(name="console") +@click.option( + "--bind", + "bind", + default="127.0.0.1:5173", + show_default=True, + help="host:port to bind. A non-loopback host (e.g. 0.0.0.0) also " + "requires --allow-remote so the proxy bridge isn't exposed openly.", +) +@click.option( + "--allow-remote", + is_flag=True, + help="Drop the loopback guard on the /proxy bridge. Only for a deployment " + "behind its own auth — a same-origin page could otherwise drive a " + "local reviewer's backends.", +) +@click.option( + "--open-browser/--no-open-browser", + default=True, + show_default=True, + help="Open the browser to the console on startup.", +) +def console(bind: str, allow_remote: bool, open_browser: bool) -> None: + """Serve the vouch web console (the React review UI) locally. + + Ships the built SPA and a same-origin /proxy bridge to your + `vouch serve --transport http` backends — one `pip install 'vouch-kb[web]'`, + no node. Add a backend from the connect dialog in the UI. + """ + from .web import _require_console_deps + + try: + _require_console_deps() + except ImportError as exc: + raise click.ClickException(str(exc)) from exc + + from .web.console import ConsoleError, resolve_console_dir, serve_console + + host, sep, port_raw = bind.partition(":") + if not sep: + raise click.ClickException(f"invalid --bind {bind!r}; expected host:port") + try: + port = int(port_raw) + except ValueError as exc: + raise click.ClickException(f"invalid port in --bind {bind!r}") from exc + host = host or "127.0.0.1" + if host not in ("127.0.0.1", "::1", "localhost") and not allow_remote: + raise click.ClickException( + f"--bind {bind} is non-loopback; pass --allow-remote to expose the " + "proxy bridge (only behind your own auth)." + ) + + url = f"http://{host}:{port}/" + if open_browser and resolve_console_dir() is not None: + import threading + import webbrowser + + threading.Timer(0.6, lambda: webbrowser.open(url)).start() + click.echo(f"vouch console → {url}") + try: + serve_console(host=host, port=port, allow_remote=allow_remote) + except ConsoleError as exc: + raise click.ClickException(str(exc)) from exc + + @cli.command(name="review-ui") @click.option( "--bind", diff --git a/src/vouch/codex_rollout.py b/src/vouch/codex_rollout.py index b23d409a..b85c9068 100644 --- a/src/vouch/codex_rollout.py +++ b/src/vouch/codex_rollout.py @@ -43,6 +43,11 @@ _ZSTD_MAGIC = b"\x28\xb5\x2f\xfd" _MAX_PROMPT_CHARS = 240 +# Upper bound on a rollout file we'll parse. Rollouts carry large tool +# outputs, so this is generous, but an unbounded read (a huge file, or a +# newline-free blob that the streaming reader would slurp whole) must not be +# able to exhaust memory. Mirrors the byte caps elsewhere (fetch, http_server). +_MAX_ROLLOUT_BYTES = 64 * 1024 * 1024 _PATCH_FILE_RE = re.compile(r"^\*\*\* (Add|Update|Delete) File: (.+)$", re.MULTILINE) _EXIT_CODE_RE = re.compile(r"exited with code (\d+)") @@ -152,14 +157,23 @@ def _iter_rollout_lines(path: Path) -> Iterator[str]: """Yield the rollout's lines one at a time, decoded as UTF-8. Streams from the file handle rather than slurping the whole file into - memory — codex rollouts can carry large tool outputs. The zstd magic - header is checked up front (compressed rollouts aren't parsed here). + memory — codex rollouts can carry large tool outputs. The size is checked + up front so an oversized rollout (or a newline-free blob that iterating the + handle would otherwise read whole) is refused instead of exhausting memory. + The zstd magic header is checked up front too (compressed rollouts aren't + parsed here). """ try: fh = path.open("rb") except OSError as e: raise CodexRolloutError(f"cannot read rollout file {path}: {e}") from e with fh: + size = os.fstat(fh.fileno()).st_size + if size > _MAX_ROLLOUT_BYTES: + raise CodexRolloutError( + f"{path.name} is too large to parse " + f"({size} bytes > {_MAX_ROLLOUT_BYTES} byte limit)" + ) if fh.read(4) == _ZSTD_MAGIC: raise CodexRolloutError( f"{path.name} is zstd-compressed; decompress it first " diff --git a/src/vouch/compile.py b/src/vouch/compile.py index 8dbf4a86..db1a168c 100644 --- a/src/vouch/compile.py +++ b/src/vouch/compile.py @@ -20,16 +20,14 @@ from __future__ import annotations -import json import re -import subprocess -import tempfile from dataclasses import dataclass, field from typing import Any import yaml from . import audit as audit_mod +from . import llm_draft from .context import _RETRACTED_CLAIM_STATUSES from .models import ProposalStatus from .proposals import ProposalError, _slugify, propose_page @@ -50,7 +48,6 @@ _WIKILINK_RE = re.compile(r"\[\[([^\]|#]+)") _CLAIM_MARKER_RE = re.compile(r"\[claim:\s*([^\]]+)\]") -_FENCE_RE = re.compile(r"^```[a-zA-Z]*\n|\n```$") class CompileError(Exception): @@ -182,52 +179,24 @@ def build_prompt(store: KBStore, *, max_pages: int) -> str: def run_llm(llm_cmd: str, prompt: str, *, timeout_seconds: float) -> str: """Run the configured LLM command with the prompt on stdin. - Runs in a throwaway temp directory: an LLM CLI that discovers per-project - hooks or MCP servers from its cwd (claude -p does) must not fire this - project's capture pipeline or connect back to this KB while compiling it. - - Explicit UTF-8 on both pipe directions — the default follows the locale - (Latin-1 on some hosts, see storage.py), which would crash on the first - em-dash in a claim or silently mojibake the drafted bodies. ``replace`` - on decode so a stray invalid byte from the LLM surfaces as a visible - replacement char in review rather than an exception. + Thin wrapper over ``llm_draft.run_llm``, translating its error into the + ``CompileError`` compile callers already handle and keeping the + "compile.llm_cmd …" wording in messages. """ - with tempfile.TemporaryDirectory(prefix="vouch-compile-") as tmp: - try: - proc = subprocess.run( - llm_cmd, shell=True, cwd=tmp, - input=prompt, capture_output=True, text=True, - encoding="utf-8", errors="replace", - timeout=timeout_seconds, - ) - except subprocess.TimeoutExpired as e: - raise CompileError( - f"compile.llm_cmd timed out after {timeout_seconds:.0f}s" - ) from e - if proc.returncode != 0: - detail = (proc.stderr or proc.stdout or "").strip()[:400] - raise CompileError(f"compile.llm_cmd failed ({proc.returncode}): {detail}") - return proc.stdout + try: + return llm_draft.run_llm( + llm_cmd, prompt, timeout_seconds=timeout_seconds, + label="compile.llm_cmd", + ) + except llm_draft.LLMDraftError as e: + raise CompileError(str(e)) from e def parse_drafts(raw: str) -> list[dict[str, Any]]: - text = raw.strip() - text = _FENCE_RE.sub("", text).strip() try: - data = json.loads(text) - except json.JSONDecodeError as e: - raise CompileError(f"compiler output is not valid JSON: {e}") from e - if not isinstance(data, list): - raise CompileError("compiler output must be a JSON array of pages") - for item in data: - if not isinstance(item, dict): - # a list of strings is a common LLM shape failure; surfacing it - # beats reporting an empty-but-successful compile. - raise CompileError( - "compiler output must be a JSON array of page objects, " - f"got element of type {type(item).__name__}" - ) - return list(data) + return llm_draft.parse_drafts(raw, noun="page") + except llm_draft.LLMDraftError as e: + raise CompileError(str(e)) from e def _draft_problem( @@ -343,6 +312,12 @@ def compile_kb( report.dropped.append({"title": title, "reason": problem}) continue survivors.append((draft, title)) + # Fix #439: fold accepted title/slug into taken_names immediately so + # a later draft in the same batch cannot collide with a just-accepted + # one. Without this, two drafts with the same title both pass + # _draft_problem's collision guard and land as duplicate proposals. + taken_names.add(title.lower()) + taken_names.add(_slugify(title)) # phase 2: wikilinks resolve against existing pages + the *surviving* # batch, to a fixpoint — dropping a draft may dangle a link in another, diff --git a/src/vouch/context.py b/src/vouch/context.py index fa7e3f16..71827be0 100644 --- a/src/vouch/context.py +++ b/src/vouch/context.py @@ -44,6 +44,7 @@ ContextItemKind = Literal["claim", "page", "entity", "relation", "source"] _VALID_BACKENDS = ("auto", "hybrid", "embedding", "fts5", "substring") +_RERANKER_CACHE: Any | None = None def _configured_backend(store: KBStore) -> str: @@ -74,6 +75,90 @@ def _configured_backend(store: KBStore) -> str: return "auto" +def _configured_rerank(store: KBStore, *, limit: int) -> tuple[bool, int]: + """Resolve the optional context rerank stage from config.yaml. + + Defaults to disabled so existing KBs keep byte-identical ordering unless + they opt in with ``retrieval.rerank.enabled: true``. ``top_k`` is the + window to reorder; by default it is the caller's context limit. + """ + try: + loaded = yaml.safe_load(store.config_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError): + return False, limit + if not isinstance(loaded, dict): + return False, limit + retrieval = loaded.get("retrieval") + if not isinstance(retrieval, dict): + return False, limit + rerank = retrieval.get("rerank") + if not isinstance(rerank, dict): + return False, limit + + enabled = rerank.get("enabled", False) + enabled = enabled if isinstance(enabled, bool) else False + + top_k = rerank.get("top_k", limit) + top_k = ( + top_k + if isinstance(top_k, int) and not isinstance(top_k, bool) and top_k > 0 + else limit + ) + return enabled, top_k + + +def _default_reranker_cached() -> Any: + global _RERANKER_CACHE + if _RERANKER_CACHE is None: + from .embeddings.rerank import default_reranker + + _RERANKER_CACHE = default_reranker() + return _RERANKER_CACHE + + +def _maybe_rerank( + store: KBStore, + *, + query: str, + hits: list[tuple[str, str, str, float]], + limit: int, +) -> list[tuple[str, str, str, float]]: + enabled, top_k = _configured_rerank(store, limit=limit) + if not enabled or not hits or top_k <= 0: + return hits + + window_size = min(top_k, len(hits)) + window = hits[:window_size] + try: + from .embeddings.rerank import rerank as do_rerank + + reranked = do_rerank( + query=query, + hits=window, + reranker=_default_reranker_cached(), + top_k=window_size, + ) + except ImportError: + return hits + + # Keep reranking as an ordering-only stage: the configured window may move, + # but it must not add/drop artifacts from the already-scoped result set. + original_by_key = {(hit[0], hit[1]): hit for hit in window} + seen: set[tuple[str, str]] = set() + ordered: list[tuple[str, str, str, float]] = [] + for hit in reranked: + key = (hit[0], hit[1]) + if key in original_by_key and key not in seen: + ordered.append(original_by_key[key]) + seen.add(key) + for hit in window: + key = (hit[0], hit[1]) + if key not in seen: + ordered.append(hit) + seen.add(key) + return ordered + hits[window_size:] + + def _retrieve( store: KBStore, query: str, @@ -101,6 +186,7 @@ def _retrieve( fused = rrf_fuse(sem, lex, limit=fetch_limit) if fused: filtered = filter_hits(store, fused, viewer, limit=limit) + filtered = _maybe_rerank(store, query=query, hits=filtered, limit=limit) return [(k, i, s, sc, "hybrid") for k, i, s, sc in filtered] # both retrievers empty -> fall through to the substring scan below. @@ -299,11 +385,6 @@ def build_context_pack( budget_clipped = 0 budget_omitted = 0 - if require_citations: - for it in items: - if it.type == "claim" and not it.citations: - uncited.append(it.id) - if max_chars is not None: total = sum(len(i.summary) for i in items) if total > max_chars: @@ -317,6 +398,14 @@ def build_context_pack( items.pop() budget_omitted += 1 + # Compute the citation gate over the items actually returned — after the + # max_chars budget has dropped tail items — so the gate never fails on (or + # reports in uncited_items) claims the consumer did not receive. + if require_citations: + uncited = [ + it.id for it in items if it.type == "claim" and not it.citations + ] + if len(items) < min_items: warnings.append(f"only {len(items)} items, minimum {min_items}") failed.append("min_items") diff --git a/src/vouch/digest.py b/src/vouch/digest.py index e5c279ad..d7bae664 100644 --- a/src/vouch/digest.py +++ b/src/vouch/digest.py @@ -197,7 +197,7 @@ def build( if str(p.metadata.get("followup_status", "")) not in _CLOSED_FOLLOWUP_STATUSES ), key=lambda r: r.due_at, - ) + )[:limit] m = compute(store, since=since, stale_after_days=stale_after_days, now=now) diff --git a/src/vouch/experts.py b/src/vouch/experts.py new file mode 100644 index 00000000..c20f409e --- /dev/null +++ b/src/vouch/experts.py @@ -0,0 +1,113 @@ +"""kb.experts - rank entities by evidence density on a topic (issue #315). + +Read-only aggregation over approved, live claims. Given a free-text topic, +return the entities carrying the most matched evidence, ranked by one of three +weightings (count / recency / citation). It never proposes, writes, or mutates +anything, makes no network or LLM call, and reads only claims already past the +review gate - so the review gate is untouched by construction. +""" + +from __future__ import annotations + +from datetime import datetime +from typing import Any + +from . import index_db +from .models import Claim, ClaimStatus, utcnow +from .salience import _substring_entity_ids +from .storage import KBStore + +# A superseded / archived / redacted claim is not live evidence and must never +# inflate an entity ranking (consistent with issue #78). +_EXCLUDED_STATUSES = frozenset( + {ClaimStatus.SUPERSEDED, ClaimStatus.ARCHIVED, ClaimStatus.REDACTED} +) +_VALID_WEIGHTS = frozenset({"count", "recency", "citation"}) +_RECENCY_HALF_LIFE_DAYS = 30.0 + + +def _claim_weight(claim: Claim, weight: str, now: datetime) -> float: + """Per-claim contribution to an entity score under the chosen weighting.""" + if weight == "recency": + ts = claim.last_confirmed_at or claim.updated_at + age_days = max(0.0, (now - ts).total_seconds() / 86400.0) + return 2.0 ** (-age_days / _RECENCY_HALF_LIFE_DAYS) + if weight == "citation": + return float(len(set(claim.evidence))) * float(claim.confidence) + return 1.0 # count + + +def rank_experts( + store: KBStore, + topic: str, + *, + limit: int = 10, + min_claims: int = 1, + weight: str = "count", +) -> list[dict[str, Any]]: + """Return entities ranked by evidence density on ``topic``. + + ``weight`` is one of ``count`` | ``recency`` | ``citation``; an unknown + value falls back to ``count`` (never raises), matching the defensive-config + style used elsewhere. Ordered by descending score with a stable tie-break + on ``entity_id``. + """ + if weight not in _VALID_WEIGHTS: + weight = "count" + + entities = store.list_entities() + by_id = {ent.id: ent for ent in entities} + topic_entity_ids = set(_substring_entity_ids(entities, topic)) + + # Candidate claims: FTS hits on the topic, plus every claim that references + # an entity whose name/alias matches the topic. + fetch = max(limit * 5, 50) + fts_claim_ids = { + cid + for kind, cid, _snip, _score in index_db.search(store.kb_dir, topic, limit=fetch) + if kind == "claim" + } + + now = utcnow() + counts: dict[str, int] = {} + citations: dict[str, set[str]] = {} + scores: dict[str, float] = {} + top_claims: dict[str, list[tuple[float, str]]] = {} + + for claim in store.list_claims(): + if claim.status in _EXCLUDED_STATUSES: + continue + matched = claim.id in fts_claim_ids or bool( + set(claim.entities) & topic_entity_ids + ) + if not matched: + continue + contrib = _claim_weight(claim, weight, now) + for eid in claim.entities: + if eid not in by_id: + continue # dangling reference - skip (graph gate should prevent) + counts[eid] = counts.get(eid, 0) + 1 + citations.setdefault(eid, set()).update(claim.evidence) + scores[eid] = scores.get(eid, 0.0) + contrib + top_claims.setdefault(eid, []).append((contrib, claim.id)) + + rows: list[dict[str, Any]] = [] + for eid, count in counts.items(): + if count < min_claims: + continue + ent = by_id[eid] + ranked = sorted(top_claims[eid], key=lambda item: (-item[0], item[1])) + rows.append( + { + "entity_id": eid, + "name": ent.name, + "type": str(ent.type), + "claim_count": count, + "citation_count": len(citations.get(eid, set())), + "score": round(scores[eid], 6), + "top_claim_ids": [cid for _w, cid in ranked[:3]], + } + ) + + rows.sort(key=lambda row: (-row["score"], -row["claim_count"], row["entity_id"])) + return rows[:limit] diff --git a/src/vouch/hooks.py b/src/vouch/hooks.py index e90c2f66..6fef47fd 100644 --- a/src/vouch/hooks.py +++ b/src/vouch/hooks.py @@ -14,6 +14,9 @@ import logging from typing import Any +import yaml + +from . import salience as salience_mod from .context import build_context_pack from .storage import KBStore @@ -46,9 +49,29 @@ def build_claude_prompt_hook(store: KBStore, stdin_text: str) -> str: prompt = str(payload.get("prompt", "")).strip() if not prompt: return "" + session_id = payload.get("session_id") + session_id = str(session_id) if session_id else None + # Feed the entity-salience reflex (#223) so repeated mentions of an + # entity within a session sharpen ranking on subsequent turns -- this + # was previously computed but never actually recorded from the hook + # path, so the reflex sat dormant for claude-code sessions (#425). + if session_id: + try: + cfg = yaml.safe_load(store.config_path.read_text(encoding="utf-8")) or {} + if not isinstance(cfg, dict): + cfg = {} + _enabled, window, _top_k = salience_mod.reflex_cfg(cfg) + salience_mod.record_query(session_id, prompt, window=window) + except Exception: + # Best-effort reflex feed: recording must never break a + # working hook (module contract -- see docstring). + _log.warning("context-hook: salience record_query failed", exc_info=True) try: pack = build_context_pack( - store, query=prompt, limit=_MAX_ITEMS, max_chars=_MAX_CHARS, + store, + query=prompt, + limit=_MAX_ITEMS, + max_chars=_MAX_CHARS, ) except Exception: _log.warning("context-hook: build_context_pack failed", exc_info=True) @@ -60,9 +83,11 @@ def build_claude_prompt_hook(store: KBStore, stdin_text: str) -> str: "Relevant knowledge from the project's vouch KB " "(approved & cited — consider it before answering):\n" + body ) - return json.dumps({ - "hookSpecificOutput": { - "hookEventName": "UserPromptSubmit", - "additionalContext": block, + return json.dumps( + { + "hookSpecificOutput": { + "hookEventName": "UserPromptSubmit", + "additionalContext": block, + } } - }) + ) diff --git a/src/vouch/index_db.py b/src/vouch/index_db.py index 146d9f52..29d6a859 100644 --- a/src/vouch/index_db.py +++ b/src/vouch/index_db.py @@ -186,6 +186,30 @@ def index_claim(conn: sqlite3.Connection, *, id: str, text: str, ) +def deindex(conn: sqlite3.Connection, *, kind: str, id: str) -> None: + """Remove every derived index row for a deleted artifact. + + FTS row for claim/page/entity (relations have no FTS table); the + embedding row for any kind (every put_* calls _embed_and_store, so an + embedding may exist for a relation too); and any provenance edge that + touches the id. prov_edges is otherwise rebuildable via + `kb.provenance_rebuild` — this keeps state.db consistent without a + full rebuild. + """ + if kind == "claim": + conn.execute("DELETE FROM claims_fts WHERE id = ?", (id,)) + elif kind == "page": + conn.execute("DELETE FROM pages_fts WHERE id = ?", (id,)) + elif kind == "entity": + conn.execute("DELETE FROM entities_fts WHERE id = ?", (id,)) + conn.execute( + "DELETE FROM embedding_index WHERE kind = ? AND id = ?", (kind, id) + ) + conn.execute( + "DELETE FROM prov_edges WHERE src_id = ? OR dst_id = ?", (id, id) + ) + + # --- provenance edges (derived cache for `vouch why/trace/impact`) -------- diff --git a/src/vouch/install_adapter.py b/src/vouch/install_adapter.py index 4fce5cd0..e6998c98 100644 --- a/src/vouch/install_adapter.py +++ b/src/vouch/install_adapter.py @@ -73,6 +73,11 @@ class InstallResult: appended: list[str] = field(default_factory=list) skipped: list[str] = field(default_factory=list) merged: list[str] = field(default_factory=list) + # files vouch wanted to install but could not (e.g. a config it can't + # faithfully re-serialize, or a malformed existing file). Kept distinct + # from ``skipped`` so "already installed" and "install failed" never look + # the same to the caller / CLI. + failed: list[str] = field(default_factory=list) @dataclass(frozen=True) @@ -235,6 +240,16 @@ def install(adapter: str, *, target: Path, tier: str = "T4") -> InstallResult: for entry in entries: src = src_root / entry.src dst = target / entry.dst + # Defense in depth: `dst` comes from the manifest. Adapters ship + # with vouch (trusted), but a `..`/absolute `dst` must never let a + # manifest write outside the target tree. `target` is already + # resolved above; resolve `dst` (normalizes `..`, even when the + # file doesn't exist yet) and require containment. + if not dst.resolve().is_relative_to(target): + raise AdapterError( + f"{adapter}: install.yaml dst {entry.dst!r} escapes the " + f"target tree ({target})" + ) if not src.is_file(): # Manifest declares a template that doesn't exist in the # adapter directory: a contributor-time bug, but surface it @@ -588,12 +603,17 @@ def _install_toml_merge( * dst exists, merge adds keys -> merge + rewrite (``merged``) * dst exists, nothing to add -> skip (``skipped``); already installed * dst exists, unparseable -> skip (``skipped``); never clobber the user + * dst exists, can't re-emit -> ``failed``; the minimal serializer can't + round-trip the merged data, so vouch is + left unwired — reported, not silently + bucketed as "already present" Rewriting re-serializes the whole file (comments and formatting are not preserved — same trade-off ``_install_json_merge`` already makes). The serialized result must survive a tomllib round-trip back to the merged data; anything the minimal serializer can't faithfully re-emit degrades - to ``skipped`` rather than risking the user's config. + to ``failed`` (the user's config is left untouched) rather than being + reported as a successful no-op. """ if not dst.exists(): dst.parent.mkdir(parents=True, exist_ok=True) @@ -605,7 +625,8 @@ def _install_toml_merge( dst_data = tomllib.loads(dst.read_text(encoding="utf-8")) src_data = tomllib.loads(src.read_text(encoding="utf-8")) except (OSError, UnicodeDecodeError, tomllib.TOMLDecodeError): - # Malformed or unreadable user file — leave it untouched. + # Malformed or unreadable user file — leave it untouched (existing + # "never clobber the user" contract; see the skipped test). result.skipped.append(rel_dst) return @@ -618,7 +639,10 @@ def _install_toml_merge( if tomllib.loads(text) != dst_data: raise ValueError("serializer round-trip mismatch") except (ValueError, tomllib.TOMLDecodeError): - result.skipped.append(rel_dst) + # The minimal serializer can't faithfully re-emit the merged config + # (e.g. a non-BMP string value, or a nan/inf float). Leave the user's + # file alone, but report it as a failure — not "already present". + result.failed.append(rel_dst) return dst.write_text(text, encoding="utf-8") diff --git a/src/vouch/jsonl_server.py b/src/vouch/jsonl_server.py index 3ada4b9a..e3785576 100644 --- a/src/vouch/jsonl_server.py +++ b/src/vouch/jsonl_server.py @@ -35,6 +35,7 @@ from . import metrics as metrics_mod from . import salience as salience_mod from . import sessions as sess_mod +from . import skills as skills_mod from . import trust as trust_mod from . import verify as verify_mod from .capabilities import capabilities as build_caps @@ -48,13 +49,14 @@ approve, expire_pending, propose_claim, + propose_delete, propose_entity, propose_page, propose_relation, reject, reject_auto_extracted, ) -from .stats import collect_stats +from .stats import collect_activity, collect_stats from .storage import ( ArtifactNotFoundError, KBNotFoundError, @@ -88,7 +90,11 @@ def _agent() -> str: def _h_capabilities(_: dict) -> dict: - return build_caps().model_dump(mode="json") + try: + publish_skills = skills_mod.publish_skills_enabled(_store()) + except Exception: + publish_skills = True + return build_caps(publish_skills=publish_skills).model_dump(mode="json") def _h_status(_: dict) -> dict: @@ -101,6 +107,20 @@ def _h_stats(p: dict) -> dict: return collect_stats(_store(), since_days=since) +def _h_activity(p: dict) -> dict: + from .scoping import viewer_from_params + + s = _store() + viewer = viewer_from_params(s, p) + return collect_activity( + s, + days=int(p.get("days", 365)), + tz_offset_minutes=int(p.get("tz_offset_minutes", 0)), + tz=p.get("tz"), + viewer=viewer, + ) + + def _h_digest(p: dict) -> dict: d = digest_mod.build( _store(), @@ -186,6 +206,20 @@ def _load_cfg(store: KBStore) -> dict: return loaded if isinstance(loaded, dict) else {} +def _h_experts(p: dict) -> dict: + from .experts import rank_experts + + return { + "experts": rank_experts( + _store(), + p["topic"], + limit=int(p.get("limit", 10)), + min_claims=int(p.get("min_claims", 1)), + weight=p.get("weight", "count"), + ) + } + + def _h_neighbors(p: dict) -> dict: from .graph import find_neighbors @@ -399,6 +433,27 @@ def _h_compile(p: dict) -> dict: return report.to_dict() +def _h_summarize_session(p: dict) -> dict: + from . import session_split + return session_split.summarize( + _store(), p["session_id"], mode=p.get("mode", "auto"), + ) + + +def _h_list_sessions(p: dict) -> dict: + from . import session_split + return {"sessions": session_split.build_session_rows(_store())} + + +def _h_session_transcript(p: dict) -> dict: + from . import transcript + session_id = p["session_id"] + agent = p.get("agent") + if agent is not None and agent not in ("claude", "codex"): + raise ValueError(f"unknown agent: {agent!r} (expected 'claude' or 'codex')") + return transcript.load_transcript(_store(), session_id, agent=agent) + + def _h_propose_entity(p: dict) -> dict: pr = propose_entity( _store(), @@ -426,6 +481,24 @@ def _h_propose_relation(p: dict) -> dict: return {"proposal_id": pr.id, "status": pr.status.value, "kind": pr.kind.value} +def _h_propose_delete(p: dict) -> dict: + pr = propose_delete( + _store(), + target_kind=p["target_kind"], + target_id=p["target_id"], + rationale=p.get("rationale"), + session_id=p.get("session_id"), + dry_run=bool(p.get("dry_run", False)), + proposed_by=_agent(), + ) + return { + "proposal_id": pr.id, + "status": pr.status.value, + "kind": pr.kind.value, + "dry_run": bool(p.get("dry_run", False)), + } + + def _h_approve(p: dict) -> dict: a = approve(_store(), p["proposal_id"], approved_by=_agent(), reason=p.get("reason")) @@ -486,6 +559,25 @@ def _h_confirm(p: dict) -> dict: if c.last_confirmed_at else None} +def _h_clear_claims(p: dict) -> dict: + from datetime import datetime + before_dt = None + if p.get("before"): + before_dt = datetime.fromisoformat(p["before"]) + to_clear = life.clear_claims( + _store(), + auto_only=p.get("auto_only", True), + before=before_dt, + actor=_agent(), + dry_run=p.get("dry_run", False), + ) + return { + "count": len(to_clear), + "claim_ids": [c.id for c in to_clear], + "dry_run": p.get("dry_run", False), + } + + def _h_cite(p: dict) -> list: out = [] for c in life.cite(_store(), p["claim_id"]): @@ -648,6 +740,20 @@ def _h_embeddings_stats(_: dict) -> dict: } +def _h_list_skills(_: dict) -> list[dict]: + return skills_mod.list_skills(_store()) + + +def _h_get_skill(p: dict) -> dict: + name = p.get("name") + if not isinstance(name, str) or not name.strip(): + raise ValueError("`name` is required") + try: + return skills_mod.get_skill(_store(), name) + except KeyError as e: + raise ValueError(str(e)) from e + + def _h_why(p: dict) -> dict: from . import provenance as prov @@ -730,9 +836,11 @@ def _h_propose_theme(p: dict) -> dict: "kb.capabilities": _h_capabilities, "kb.status": _h_status, "kb.stats": _h_stats, + "kb.activity": _h_activity, "kb.digest": _h_digest, "kb.search": _h_search, "kb.neighbors": _h_neighbors, + "kb.experts": _h_experts, "kb.context": _h_context, "kb.synthesize": _h_synthesize, "kb.read_page": _h_read_page, @@ -752,8 +860,12 @@ def _h_propose_theme(p: dict) -> dict: "kb.propose_claim": _h_propose_claim, "kb.propose_page": _h_propose_page, "kb.compile": _h_compile, + "kb.summarize_session": _h_summarize_session, + "kb.list_sessions": _h_list_sessions, + "kb.session_transcript": _h_session_transcript, "kb.propose_entity": _h_propose_entity, "kb.propose_relation": _h_propose_relation, + "kb.propose_delete": _h_propose_delete, "kb.approve": _h_approve, "kb.reject": _h_reject, "kb.reject_extracted": _h_reject_extracted, @@ -762,6 +874,7 @@ def _h_propose_theme(p: dict) -> dict: "kb.contradict": _h_contradict, "kb.archive": _h_archive, "kb.confirm": _h_confirm, + "kb.clear_claims": _h_clear_claims, "kb.cite": _h_cite, "kb.source_verify": _h_source_verify, "kb.session_start": _h_session_start, @@ -787,6 +900,8 @@ def _h_propose_theme(p: dict) -> dict: "kb.provenance_rebuild": _h_provenance_rebuild, "kb.detect_themes": _h_detect_themes, "kb.propose_theme": _h_propose_theme, + "kb.list_skills": _h_list_skills, + "kb.get_skill": _h_get_skill, } @@ -807,6 +922,11 @@ def handle_request(envelope: dict) -> dict: "ok": True, "result": trust_mod.finish_kb_result(result), } + except skills_mod.SkillsDisabledError as e: + return { + "id": req_id, "ok": False, + "error": {"code": "permission_denied", "message": str(e)}, + } except KeyError as e: return { "id": req_id, "ok": False, diff --git a/src/vouch/lifecycle.py b/src/vouch/lifecycle.py index fadbb072..b7582d73 100644 --- a/src/vouch/lifecycle.py +++ b/src/vouch/lifecycle.py @@ -76,6 +76,8 @@ def contradict( actor: str, ) -> tuple[Claim, Claim, Relation]: """Record that two claims contradict each other (symmetric).""" + if claim_a == claim_b: + raise LifecycleError("a claim cannot contradict itself") a = store.get_claim(claim_a) b = store.get_claim(claim_b) a.contradicts = sorted({*a.contradicts, b.id}) @@ -127,6 +129,73 @@ def confirm(store: KBStore, *, claim_id: str, actor: str) -> Claim: return claim +def clear_claims( + store: KBStore, + *, + auto_only: bool = True, + before: datetime | None = None, + actor: str, + dry_run: bool = False, +) -> list[Claim]: + """Clear auto-approved claims, optionally filtered by date range. + + Filters claims by auto-approval status and/or created_at timestamp, + then archives (not deletes) the matching claims. All operations are + audited. + + Args: + store: Knowledge base store + auto_only: If True, only clear auto-approved claims (auto_approved=True). + If False, clear all claims matching date filter. + before: If set, only clear claims created before this datetime. + actor: Who is performing the operation. + dry_run: If True, don't write changes, just return what would be cleared. + + Returns: + List of claims that were (or would be) archived. + """ + all_claims = store.list_claims() + to_clear: list[Claim] = [] + + for claim in all_claims: + # Skip if already archived or status doesn't allow clearing + if claim.status == ClaimStatus.ARCHIVED: + continue + + # Filter by auto-approval status + if auto_only and not claim.auto_approved: + continue + + # Filter by date range + if before and claim.created_at >= before: + continue + + to_clear.append(claim) + + # Apply the clear operation (archive the claims) + if not dry_run: + for claim in to_clear: + claim.status = ClaimStatus.ARCHIVED + claim.updated_at = datetime.now(UTC) + store.update_claim(claim) + + # Log the bulk operation + if to_clear: + audit.log_event( + store.kb_dir, + event="claim.bulk_clear", + actor=actor, + object_ids=[c.id for c in to_clear], + data={ + "count": len(to_clear), + "auto_only": auto_only, + "before": before.isoformat() if before else None, + }, + ) + + return to_clear + + def cite(store: KBStore, claim_id: str) -> list[Evidence | dict]: """Return resolved citations for a claim. diff --git a/src/vouch/llm_draft.py b/src/vouch/llm_draft.py new file mode 100644 index 00000000..3c3f1943 --- /dev/null +++ b/src/vouch/llm_draft.py @@ -0,0 +1,96 @@ +"""Shared LLM drafting plumbing for the review-gated compilers. + +Both `compile.py` (approved claims -> topic pages) and `session_split.py` +(session observations -> topical session pages) hand a deployment-configured +LLM command a prompt on stdin and parse a JSON array of drafts back. This +module is the shared subprocess + parse layer; each caller keeps its own +domain validation (compile verifies claim citations; session_split forces the +session page type). +""" + +from __future__ import annotations + +import json +import re +import subprocess +import tempfile +from typing import Any + +_FENCE_RE = re.compile(r"^```[a-zA-Z]*\n|\n```$") + + +class LLMDraftError(Exception): + """The LLM command could not run, or returned unusable output.""" + + +def run_llm( + llm_cmd: str, + prompt: str, + *, + timeout_seconds: float, + label: str = "llm_cmd", +) -> str: + """Run `llm_cmd` with `prompt` on stdin, in a throwaway temp cwd. + + `label` names the command in error messages so callers keep their own + config-key wording (e.g. "compile.llm_cmd"). Runs in a temp dir so an LLM + CLI that discovers per-project hooks/MCP from its cwd does not fire this + project's own pipeline while summarizing it. UTF-8 is forced on both pipe + directions — the default follows the locale (Latin-1 on some hosts), which + would crash on the first em-dash; `errors="replace"` surfaces a stray + invalid byte as a visible replacement char in review, not an exception. + """ + with tempfile.TemporaryDirectory(prefix="vouch-llm-") as tmp: + try: + proc = subprocess.run( + llm_cmd, shell=True, cwd=tmp, + input=prompt, capture_output=True, text=True, + encoding="utf-8", errors="replace", + timeout=timeout_seconds, + ) + except subprocess.TimeoutExpired as e: + raise LLMDraftError(f"{label} timed out after {timeout_seconds:.0f}s") from e + if proc.returncode != 0: + detail = (proc.stderr or proc.stdout or "").strip()[:400] + raise LLMDraftError(f"{label} failed ({proc.returncode}): {detail}") + return proc.stdout + + +def parse_object(raw: str, *, noun: str = "answer") -> dict[str, Any]: + """Parse LLM stdout into a single JSON object. + + The object-shaped sibling of `parse_drafts`: same fence stripping, same + LLMDraftError contract, for callers whose LLM returns one object (e.g. + synthesize's `{"answer": …, "gaps": […]}`) rather than an array of drafts. + """ + text = _FENCE_RE.sub("", raw.strip()).strip() + try: + data = json.loads(text) + except json.JSONDecodeError as e: + raise LLMDraftError(f"{noun} output is not valid JSON: {e}") from e + if not isinstance(data, dict): + raise LLMDraftError(f"{noun} output must be a single JSON object") + return data + + +def parse_drafts(raw: str, *, noun: str = "page") -> list[dict[str, Any]]: + """Parse LLM stdout into a list of draft dicts. + + Strips a single markdown code fence if present. `noun` tunes error wording + ("page" -> "JSON array of pages"). Raises LLMDraftError on any shape + failure so callers can surface it as a clean, caller-visible message. + """ + text = _FENCE_RE.sub("", raw.strip()).strip() + try: + data = json.loads(text) + except json.JSONDecodeError as e: + raise LLMDraftError(f"compiler output is not valid JSON: {e}") from e + if not isinstance(data, list): + raise LLMDraftError(f"compiler output must be a JSON array of {noun}s") + for item in data: + if not isinstance(item, dict): + raise LLMDraftError( + f"compiler output must be a JSON array of {noun} objects, " + f"got element of type {type(item).__name__}" + ) + return list(data) diff --git a/src/vouch/models.py b/src/vouch/models.py index 5159d8ff..7946f30f 100644 --- a/src/vouch/models.py +++ b/src/vouch/models.py @@ -267,6 +267,8 @@ def _text_non_empty(cls, v: str) -> str: updated_at: datetime = Field(default_factory=utcnow) last_confirmed_at: datetime | None = None approved_by: str | None = None # vouch: review-gate audit + proposed_by: str | None = None # vouch: tracks who proposed this claim + auto_approved: bool = False # vouch: true if approved by proposer (trusted-agent mode) @field_validator("scope", mode="before") @classmethod @@ -394,6 +396,7 @@ class ProposalKind(StrEnum): PAGE = "page" ENTITY = "entity" RELATION = "relation" + DELETE = "delete" class ProposalStatus(StrEnum): @@ -493,6 +496,13 @@ class Capabilities(BaseModel): default_factory=list, description="OpenClaw context engines exposed (see openclaw.plugin.json)", ) + mcp: dict[str, Any] = Field( + default_factory=lambda: {"publish_skills": True}, + description=( + "mcp surface flags mirrored from config.yaml `mcp:` block. " + "`publish_skills` gates kb.list_skills / kb.get_skill." + ), + ) host_compat: dict[str, Any] = Field( default_factory=dict, description=( diff --git a/src/vouch/proposals.py b/src/vouch/proposals.py index 876c13e1..e66c1298 100644 --- a/src/vouch/proposals.py +++ b/src/vouch/proposals.py @@ -316,6 +316,53 @@ def propose_relation( ) +def propose_delete( + store: KBStore, + *, + target_kind: str, + target_id: str, + proposed_by: str, + rationale: str | None = None, + session_id: str | None = None, + dry_run: bool = False, +) -> Proposal: + """File a review-gated request to hard-delete a durable artifact. + + Blocked (at propose time, re-checked at approve) if the target is still + referenced by another artifact — the maintainer must supersede or remove + the referrers first. The full artifact is snapshotted into the payload so + the decided proposal and audit event record exactly what was removed. + """ + if target_kind not in _DELETE_KINDS: + raise ProposalError( + f"unknown target_kind {target_kind!r}; expected one of " + f"{sorted(_DELETE_KINDS)}" + ) + getter = getattr(store, _DELETE_GETTERS[target_kind]) + try: + artifact = getter(target_id) + except ArtifactNotFoundError as e: + raise ProposalError(f"unknown {target_kind} id: {target_id}") from e + refs = referenced_by(store, target_kind, target_id) + if refs: + hint = " (supersede it instead?)" if target_kind == "claim" else "" + raise ProposalError( + f"cannot delete {target_kind} {target_id}: referenced by " + + ", ".join(refs) + + hint + ) + payload = { + "target_kind": target_kind, + "id": target_id, + "snapshot": artifact.model_dump(mode="json"), + } + return _file_proposal( + store, kind=ProposalKind.DELETE, payload=payload, + proposed_by=proposed_by, session_id=session_id, + rationale=rationale, dry_run=dry_run, + ) + + # --- decisions ------------------------------------------------------------ @@ -412,6 +459,22 @@ def _payload_block_reason(store: KBStore, proposal: Proposal) -> str | None: Entity(**payload) except (ValidationError, TypeError) as e: return f"invalid entity payload: {e}" + elif proposal.kind == ProposalKind.DELETE: + target_kind = str(payload.get("target_kind", "")) + target_id = str(payload.get("id", "")) + if target_kind not in _DELETE_KINDS: + return f"invalid delete target_kind: {target_kind!r}" + getter = getattr(store, _DELETE_GETTERS[target_kind]) + try: + getter(target_id) + except ArtifactNotFoundError: + return None # already gone → idempotent approve is fine + refs = referenced_by(store, target_kind, target_id) + if refs: + return ( + f"cannot delete {target_kind} {target_id}: referenced by " + + ", ".join(refs) + ) return None @@ -457,11 +520,17 @@ def approve( # Exception: PAGE proposals may legitimately target an existing page when # filed by vault_to_kb (vault edit flow) — the approve path handles that # via update_page rather than put_page. - if proposal.kind != ProposalKind.PAGE: + if proposal.kind not in (ProposalKind.PAGE, ProposalKind.DELETE): _ensure_no_existing_artifact(store, proposal.kind, payload["id"]) result: Claim | Page | Entity | Relation if proposal.kind == ProposalKind.CLAIM: - claim = Claim(approved_by=approved_by, **payload) + is_auto_approved = approved_by == proposal.proposed_by + claim = Claim( + approved_by=approved_by, + proposed_by=proposal.proposed_by, + auto_approved=is_auto_approved, + **payload + ) store.put_claim(claim) with index_db.open_db(store.kb_dir) as conn: index_db.index_claim( @@ -513,6 +582,8 @@ def approve( type=entity.type.value, aliases=entity.aliases, ) result = entity + elif proposal.kind == ProposalKind.DELETE: + result = _approve_delete(store, proposal, approved_by=approved_by) else: # RELATION rel = Relation(**payload) store.put_relation(rel) @@ -682,6 +753,131 @@ def expire_pending( } +_DELETE_KINDS = {"claim", "page", "entity", "relation"} + +_DELETE_GETTERS = { + "claim": "get_claim", + "page": "get_page", + "entity": "get_entity", + "relation": "get_relation", +} + + +def referenced_by(store: KBStore, target_kind: str, target_id: str) -> list[str]: + """Inbound referrers to `target_id` — the "block if referenced" gate. + + Returns human-readable descriptions of artifacts that point AT the + target. Only inbound refs count; outbound refs (what the target itself + points at) are never returned, because deleting the holder simply drops + its own pointers. An empty list means the artifact is safe to delete. + """ + if target_kind not in _DELETE_KINDS: + raise ProposalError( + f"unknown target_kind {target_kind!r}; expected one of " + f"{sorted(_DELETE_KINDS)}" + ) + refs: list[str] = [] + if target_kind == "claim": + for page in store.list_pages(): + if target_id in page.claims: + refs.append(f"page {page.id!r}") + # relation endpoints are bare ids without a kind tag; a same-slug + # artifact of a different kind could match here. acceptable given the + # slug-collision caveat in the spec's "out of scope". + for rel in store.list_relations(): + if target_id in (rel.source, rel.target): + refs.append(f"relation {rel.id!r}") + for claim in store.list_claims(): + if claim.id == target_id: + continue + if ( + target_id in claim.supersedes + or claim.superseded_by == target_id + or target_id in claim.contradicts + ): + refs.append(f"claim {claim.id!r}") + elif target_kind == "page": + for rel in store.list_relations(): + if target_id in (rel.source, rel.target): + refs.append(f"relation {rel.id!r}") + elif target_kind == "entity": + for claim in store.list_claims(): + if target_id in claim.entities: + refs.append(f"claim {claim.id!r}") + for page in store.list_pages(): + if target_id in page.entities: + refs.append(f"page {page.id!r}") + for rel in store.list_relations(): + if target_id in (rel.source, rel.target): + refs.append(f"relation {rel.id!r}") + # target_kind == "relation": edges have no inbound refs → refs stays empty + return refs + + +def _reconstruct_deleted( + target_kind: str, snapshot: dict[str, Any] +) -> Claim | Page | Entity | Relation: + """Rebuild a typed model from a delete proposal's snapshot. + + Used only on the idempotent path (artifact already gone) so the approve + surfaces still receive a `{kind, id}` result. + """ + if target_kind == "claim": + return Claim(**snapshot) + if target_kind == "page": + return Page(**snapshot) + if target_kind == "entity": + return Entity(**snapshot) + return Relation(**snapshot) + + +def _approve_delete( + store: KBStore, proposal: Proposal, *, approved_by: str +) -> Claim | Page | Entity | Relation: + """Execute an approved DELETE proposal: remove the artifact + index rows. + + Re-checks references at approve time (they may have appeared since the + proposal was filed). Idempotent: if the artifact is already gone, finalize + the proposal without erroring. + """ + payload = proposal.payload + target_kind = str(payload["target_kind"]) + target_id = str(payload["id"]) + snapshot = dict(payload.get("snapshot") or {}) + getter = getattr(store, _DELETE_GETTERS[target_kind]) + try: + artifact = getter(target_id) + except ArtifactNotFoundError: + # Idempotent already-gone path (crash-retry between the file unlink and + # move_proposal_to_decided). The file is gone but the derived index rows + # may not be — a crash between deleter() and deindex() below would leave + # stale fts/embedding/prov rows and keep the deleted artifact searchable. + # deindex is a no-op when the rows are already absent, so run it here too + # to converge the index. The per-kind {kind}.delete audit event is + # intentionally NOT re-emitted on this path: the snapshot is preserved in + # the decided/ proposal, the shared approve() tail still records + # proposal.delete.approve, and re-emitting would double-log if the crash + # landed after the original audit call. + with index_db.open_db(store.kb_dir) as conn: + index_db.deindex(conn, kind=target_kind, id=target_id) + return _reconstruct_deleted(target_kind, snapshot) + refs = referenced_by(store, target_kind, target_id) + if refs: + raise ProposalError( + f"cannot delete {target_kind} {target_id}: still referenced by " + + ", ".join(refs) + ) + deleter = getattr(store, f"delete_{target_kind}") + deleter(target_id) + with index_db.open_db(store.kb_dir) as conn: + index_db.deindex(conn, kind=target_kind, id=target_id) + audit.log_event( + store.kb_dir, event=f"{target_kind}.delete", actor=approved_by, + object_ids=[target_id], data={"snapshot": snapshot}, reversible=False, + ) + return artifact + + def _ensure_no_existing_artifact( store: KBStore, kind: ProposalKind, artifact_id: str ) -> None: diff --git a/src/vouch/server.py b/src/vouch/server.py index f2a4482e..7f4bfde5 100644 --- a/src/vouch/server.py +++ b/src/vouch/server.py @@ -26,6 +26,7 @@ from . import metrics as metrics_mod from . import salience as salience_mod from . import sessions as sess_mod +from . import skills as skills_mod from . import trust as trust_mod from . import verify as verify_mod from .capabilities import capabilities as build_caps @@ -39,6 +40,7 @@ approve, expire_pending, propose_claim, + propose_delete, propose_entity, propose_page, propose_relation, @@ -46,7 +48,7 @@ reject_auto_extracted, ) from .scoping import filter_hits, scoped_fetch_limit, viewer_from -from .stats import collect_stats +from .stats import collect_activity, collect_stats from .storage import ( ArtifactNotFoundError, KBNotFoundError, @@ -77,7 +79,11 @@ def _agent() -> str: @mcp.tool() def kb_capabilities() -> dict[str, Any]: """Return the protocol capabilities of this server.""" - return build_caps().model_dump(mode="json") + try: + publish_skills = skills_mod.publish_skills_enabled(_store()) + except Exception: + publish_skills = True + return build_caps(publish_skills=publish_skills).model_dump(mode="json") @mcp.tool() @@ -96,6 +102,64 @@ def kb_stats(*, days: int = 30) -> dict[str, Any]: return collect_stats(_store(), since_days=since) +@mcp.tool() +def kb_list_skills() -> list[dict[str, Any]]: + """Enumerate every Claude Code skill / slash command visible to vouch. + + Scans, in priority order: + 1. ``/.claude/skills//SKILL.md`` — project-local skills + 2. ``/.claude/commands/.md`` — project-local commands + 3. ``~/.claude/skills//SKILL.md`` — user-global skills + 4. ``~/.claude/commands/.md`` — user-global commands + + Project entries override user ones with the same name. Returns + ``[{name, description, scope, kind, path}]``. Returns an empty list + when ``mcp.publish_skills`` is ``false`` in config.yaml. + """ + return skills_mod.list_skills(_store()) + + +@mcp.tool() +def kb_get_skill(name: str) -> dict[str, Any]: + """Return the full markdown body of a named skill / slash command. + + Errors with ``permission_denied`` when ``mcp.publish_skills`` is + ``false``, and ``not_found`` when the name isn't in the catalogue. + """ + try: + return skills_mod.get_skill(_store(), name) + except skills_mod.SkillsDisabledError as e: + raise PermissionError(str(e)) from e + except KeyError as e: + raise ValueError(str(e)) from e + + +@mcp.tool() +def kb_activity( + *, + days: int = 365, + tz_offset_minutes: int = 0, + tz: str | None = None, + project: str | None = None, + agent: str | None = None, +) -> dict[str, Any]: + """Audit activity buckets for dashboards: per-day counts, hour-of-week + matrix, actor and event histograms. Scope-filtered like kb.audit. + + days: window in local calendar days; 0 means all-time. + tz: IANA zone for local-time bucketing; falls back to tz_offset_minutes. + """ + store = _store() + viewer = viewer_from( + config_path=store.config_path, + project=project, + agent=agent, + ) + return collect_activity( + store, days=days, tz_offset_minutes=tz_offset_minutes, tz=tz, viewer=viewer, + ) + + @mcp.tool() def kb_digest( *, @@ -209,6 +273,23 @@ def _load_cfg(store: KBStore) -> dict[str, Any]: return loaded if isinstance(loaded, dict) else {} +@mcp.tool() +def kb_experts( + topic: str, + limit: int = 10, + min_claims: int = 1, + weight: str = "count", +) -> dict[str, Any]: + """Rank entities by evidence density on a topic (read-only).""" + from .experts import rank_experts + + return { + "experts": rank_experts( + _store(), topic, limit=limit, min_claims=min_claims, weight=weight + ) + } + + @mcp.tool() def kb_neighbors( node_id: str, @@ -265,14 +346,20 @@ def kb_synthesize( query: str, depth: int = 3, max_chars: int = 4000, + llm: bool = False, ) -> dict[str, Any]: - """Answer a query from approved claims only, with inline `[claim_id]` - citations, an explicit gaps block, and a synthesis_confidence grade. + """Answer a query from the review-gated KB, with inline `[id]` citations, + an explicit gaps block, and a synthesis_confidence grade. Unlike `kb_context` (a ranked list), this returns prose where every - sentence is traceable to an approved claim. + sentence is traceable to a source. Deterministic by default (approved + claims only); `llm=True` drafts the answer with the deployment-configured + LLM (compile.llm_cmd) grounded in pages and approved claims — citations + are still verified mechanically, and the call is synchronous. """ - return synthesize(_store(), query=query, depth=depth, max_chars=max_chars) + return synthesize( + _store(), query=query, depth=depth, max_chars=max_chars, llm=llm, + ) @mcp.tool() @@ -527,6 +614,49 @@ def kb_compile( return report.to_dict() +@mcp.tool() +def kb_summarize_session( + session_id: str, + mode: str = "auto", +) -> dict[str, Any]: + """Summarize a captured session into PENDING page proposals. + + Reads the host-neutral observation buffer for `session_id` and files either + one mechanical rollup page (small sessions) or several LLM-drafted topical + `session` pages (large sessions). `mode` is "auto" | "split" | "mechanical". + Long-running when it splits (the LLM call is synchronous). Never approves. + """ + from . import session_split + return session_split.summarize(_store(), session_id, mode=mode) + + +@mcp.tool() +def kb_list_sessions() -> dict[str, Any]: + """List captured sessions in the review pipeline: open buffers awaiting a + summary, and filed summary proposals awaiting review. + + Read-only. Each row: session_id, stage ("buffer" | "pending"), proposal_id, + kind, title, summarized, observations, last_activity. + """ + from . import session_split + return {"sessions": session_split.build_session_rows(_store())} + + +@mcp.tool() +def kb_session_transcript(session_id: str, agent: str | None = None) -> dict[str, Any]: + """Render a captured session's full transcript from its raw agent JSONL. + + Read-only. Locates the raw Claude Code / Codex file on disk and normalizes + it into message blocks (text, thinking, tool_use with paired results). + ``agent`` restricts the search ("claude" | "codex"); omit to try both. + Degrades to compact capture observations when the raw file is unavailable. + """ + from . import transcript + if agent is not None and agent not in ("claude", "codex"): + raise ValueError(f"unknown agent: {agent!r} (expected 'claude' or 'codex')") + return transcript.load_transcript(_store(), session_id, agent=agent) + + @mcp.tool() def kb_propose_entity( name: str, @@ -574,6 +704,27 @@ def kb_propose_relation( return _proposal_response(pr, dry_run) +@mcp.tool() +def kb_propose_delete( + target_kind: str, target_id: str, rationale: str | None = None, + session_id: str | None = None, dry_run: bool = False, +) -> dict[str, Any]: + """Propose hard-deleting a durable artifact (claim/page/entity/relation). + + Files a PENDING delete request that a *different* reviewer approves via + kb.approve. Refused if the target is still referenced by another artifact. + """ + try: + pr = propose_delete( + _store(), target_kind=target_kind, target_id=target_id, + proposed_by=_agent(), rationale=rationale, + session_id=session_id, dry_run=dry_run, + ) + except (ProposalError, ArtifactNotFoundError, ValueError) as e: + raise ValueError(str(e)) from e + return _proposal_response(pr, dry_run) + + def _proposal_response(result, dry_run: bool) -> dict[str, Any]: pr = result.proposal if hasattr(result, "proposal") else result out: dict[str, Any] = { @@ -684,6 +835,42 @@ def kb_confirm(claim_id: str) -> dict[str, Any]: return {"id": c.id, "last_confirmed_at": c.last_confirmed_at} +@mcp.tool() +def kb_clear_claims( + auto_only: bool = True, + before: str | None = None, + dry_run: bool = False, +) -> dict[str, Any]: + """Clear auto-approved claims, optionally filtered by date range. + + Args: + auto_only: If True, only clear auto-approved claims. + before: If set, only clear claims created before this ISO 8601 date. + dry_run: If True, preview without making changes. + + Returns: + Dictionary with count of cleared claims and their IDs. + """ + from datetime import datetime + + before_dt = None + if before: + before_dt = datetime.fromisoformat(before) + + to_clear = life.clear_claims( + _store(), + auto_only=auto_only, + before=before_dt, + actor=_agent(), + dry_run=dry_run, + ) + return { + "count": len(to_clear), + "claim_ids": [c.id for c in to_clear], + "dry_run": dry_run, + } + + @mcp.tool() def kb_cite(claim_id: str) -> list[dict[str, Any]]: """Return resolved citations (sources or evidence records) backing a claim.""" diff --git a/src/vouch/session_split.py b/src/vouch/session_split.py new file mode 100644 index 00000000..c14187c4 --- /dev/null +++ b/src/vouch/session_split.py @@ -0,0 +1,537 @@ +"""Summarize a session's observation buffer into review-gated pages. + +Host-blind: reads only the normalized observation buffer +(`.vouch/captures/.jsonl`) that every host adapter writes via +`capture.observe`, never a host transcript. Small sessions get one mechanical +rollup page (reusing `capture.build_summary_body`); large sessions get an LLM +topical split into several `type: session` pages. Every page is a PENDING +proposal — `approve()` is never called. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any + +import yaml + +from . import audit as audit_mod +from . import capture, llm_draft +from . import compile as compile_mod +from .llm_draft import LLMDraftError +from .models import ProposalStatus +from .proposals import _slugify, propose_page, reject +from .storage import KBStore + +logger = logging.getLogger(__name__) + +SPLIT_ACTOR = "session-split" + +DEFAULT_THRESHOLD_OBSERVATIONS = 40 +DEFAULT_MAX_PAGES = 6 +DEFAULT_TIMEOUT_SECONDS = 180.0 +DEFAULT_MAX_INPUT_CHARS = 60000 + + +class SplitConfigError(Exception): + """The split cannot run (no resolvable llm_cmd).""" + + +@dataclass(frozen=True) +class SplitConfig: + enabled: bool = True + llm_cmd: str | None = None + threshold_observations: int = DEFAULT_THRESHOLD_OBSERVATIONS + max_pages: int = DEFAULT_MAX_PAGES + timeout_seconds: float = DEFAULT_TIMEOUT_SECONDS + max_input_chars: int = DEFAULT_MAX_INPUT_CHARS + + +def _coerce(value: Any, default: Any, cast: Any) -> Any: + try: + return cast(value) + except (TypeError, ValueError): + return default + + +def load_split_config(store: KBStore) -> SplitConfig: + """Read `capture.split` from config.yaml; fall back to defaults.""" + try: + loaded = yaml.safe_load(store.config_path.read_text(encoding="utf-8")) + except (OSError, yaml.YAMLError): + return SplitConfig() + if not isinstance(loaded, dict): + return SplitConfig() + cap = loaded.get("capture") + raw = cap.get("split") if isinstance(cap, dict) else None + if not isinstance(raw, dict): + return SplitConfig() + llm_cmd = raw.get("llm_cmd") + return SplitConfig( + enabled=bool(raw.get("enabled", True)), + llm_cmd=str(llm_cmd) if llm_cmd else None, + threshold_observations=_coerce( + raw.get("threshold_observations", DEFAULT_THRESHOLD_OBSERVATIONS), + DEFAULT_THRESHOLD_OBSERVATIONS, int), + max_pages=_coerce(raw.get("max_pages", DEFAULT_MAX_PAGES), DEFAULT_MAX_PAGES, int), + timeout_seconds=_coerce( + raw.get("timeout_seconds", DEFAULT_TIMEOUT_SECONDS), + DEFAULT_TIMEOUT_SECONDS, float), + max_input_chars=_coerce( + raw.get("max_input_chars", DEFAULT_MAX_INPUT_CHARS), + DEFAULT_MAX_INPUT_CHARS, int), + ) + + +def summarize( + store: KBStore, + session_id: str, + *, + intent: str | None = None, + cwd: Path | None = None, + project: str | None = None, + generated_at: str | None = None, + mode: str = "auto", + config: capture.CaptureConfig | None = None, +) -> dict[str, Any]: + """Roll a session buffer into PENDING page proposals. Never approves. + + `mode`: "auto" (size gate decides), "split" (force LLM), or "mechanical" + (force the single rollup). The buffer is deleted only after a page is + filed (or an explicit below-min skip), so a crash mid-run leaves it intact + for the next `finalize-all` sweep to retry. + """ + cfg = config or capture.load_config(store) + path = capture.buffer_path(store, session_id) + observations = capture._read_observations(path) + if not cfg.enabled: + return {"captured": len(observations), "summary_proposal_id": None, + "summary_proposal_ids": [], "mode": "skipped", "skipped": "disabled", + "session_id": session_id, "summarized": False, "proposal_id": None} + if cwd is not None: + changed_files, git_stat = capture._git_changes(cwd) + else: + changed_files, git_stat = [], "" + total = len(observations) + len(changed_files) + if total < cfg.min_observations: + # The buffer is empty/gone. If a mechanical summary was already filed + # for this session, the intent (e.g. the review console's Summarize) is + # to narrate that filed record with the LLM, not to re-read the buffer. + if mode != "mechanical": + renarrated = _try_renarrate(store, session_id, split_cfg=load_split_config(store)) + if renarrated is not None: + if path.exists(): + path.unlink() + return renarrated + if path.exists(): + path.unlink() + reason = "below-min" if observations else "no-pending-summary-for-session" + return {"captured": total, "summary_proposal_id": None, + "summary_proposal_ids": [], "mode": "skipped", "skipped": reason, + "session_id": session_id, "summarized": False, "proposal_id": None} + + split_cfg = load_split_config(store) + want_split = mode == "split" or ( + mode == "auto" and split_cfg.enabled and total >= split_cfg.threshold_observations + ) + if mode != "mechanical" and want_split: + try: + ids, dropped, truncated = _propose_split( + store, session_id, observations, changed_files, git_stat, + intent=intent, split_cfg=split_cfg, + ) + if ids: + if path.exists(): + path.unlink() + return {"captured": total, "summary_proposal_id": ids[0], + "summary_proposal_ids": ids, "mode": "split", + "dropped": dropped, "truncated": truncated, + "session_id": session_id, "summarized": True, + "proposal_id": ids[0]} + logger.warning( + "session_split: no valid drafts for %s; falling back to mechanical", + session_id, + ) + except (LLMDraftError, SplitConfigError) as e: + logger.warning( + "session_split: llm split failed for %s (%s); falling back", session_id, e + ) + + pid = _propose_mechanical( + store, session_id, observations, changed_files, git_stat, + project=project, generated_at=generated_at, intent=intent, + ) + if path.exists(): + path.unlink() + final_mode = "fallback" if (mode != "mechanical" and want_split) else "mechanical" + result: dict[str, Any] = { + "captured": total, "summary_proposal_id": pid, + "summary_proposal_ids": [pid], "mode": final_mode, + "session_id": session_id, "summarized": final_mode == "mechanical", + "proposal_id": pid, + } + if final_mode == "fallback": + # the LLM was attempted and fell back; the mechanical page is a backstop, + # but surface the failure so the UI can prompt a retry / config fix. + result["skipped"] = "llm-failed" + return result + + +def _propose_mechanical( + store: KBStore, + session_id: str, + observations: list[dict[str, Any]], + changed_files: list[str], + git_stat: str, + *, + project: str | None, + generated_at: str | None, + intent: str | None, +) -> str: + """File the single mechanical rollup page, exactly as capture did before.""" + title, body = capture.build_summary_body( + session_id, observations, changed_files, git_stat, + project=project, generated_at=generated_at, first_prompt=intent, + ) + proposal = propose_page( + store, title=title, body=body, + page_type=capture.CAPTURE_PAGE_TYPE, + proposed_by=capture.CAPTURE_ACTOR, + session_id=session_id, + rationale="auto-captured session summary", + ) + return proposal.id + + +def _propose_split( + store: KBStore, + session_id: str, + observations: list[dict[str, Any]], + changed_files: list[str], + git_stat: str, + *, + intent: str | None, + split_cfg: SplitConfig, +) -> tuple[list[str], list[dict[str, Any]], bool]: + cmd = split_cfg.llm_cmd or compile_mod.load_config(store).llm_cmd + if not cmd: + raise SplitConfigError( + "capture.split.llm_cmd is not configured (and compile.llm_cmd is unset)" + ) + prompt, truncated = build_split_prompt( + store, observations, changed_files, git_stat, + intent=intent, max_pages=split_cfg.max_pages, + max_input_chars=split_cfg.max_input_chars, + ) + raw = llm_draft.run_llm( + cmd, prompt, timeout_seconds=split_cfg.timeout_seconds, + label="capture.split.llm_cmd", + ) + drafts = llm_draft.parse_drafts(raw, noun="page") + ids, dropped = _file_drafts(store, session_id, drafts, split_cfg.max_pages) + _audit_split(store, session_id, ids, dropped, len(observations), truncated) + return ids, dropped, truncated + + +def _render_obs(obs: dict[str, Any]) -> str: + tool = str(obs.get("tool", "")).strip() + summary = str(obs.get("summary", "")).strip() + files = obs.get("files") or [] + line = f"[{tool}] {summary}" if tool else summary + if files: + line += f" (files: {', '.join(str(f) for f in files[:5])})" + return line + + +def build_split_prompt( + store: KBStore, + observations: list[dict[str, Any]], + changed_files: list[str], + git_stat: str, + *, + intent: str | None, + max_pages: int, + max_input_chars: int, +) -> tuple[str, bool]: + """Assemble the host-neutral topical-split prompt. Returns (prompt, truncated). + + `tool` labels are opaque — the model clusters on the `summary` prose, so any + host's tool vocabulary works. If the rendered activity exceeds + `max_input_chars`, keep the most-recent observations that fit and prepend an + explicit elision note (no silent cap). + """ + obs_lines = [_render_obs(o) for o in observations] + truncated = False + if sum(len(x) + 1 for x in obs_lines) > max_input_chars: + truncated = True + kept: list[str] = [] + size = 0 + for line in reversed(obs_lines): + if size + len(line) + 1 > max_input_chars: + break + kept.append(line) + size += len(line) + 1 + obs_lines = list(reversed(kept)) + elided = len(observations) - len(obs_lines) + obs_lines.insert(0, f"(... {elided} older observations elided ...)") + + lines: list[str] = [ + "You are the session historian for this project's knowledge base. You", + "summarize one work session into a small set of durable, human-readable", + "session records — one per distinct thread of work.", + "", + ] + if intent: + lines += ["SESSION INTENT:", f" {intent}", ""] + lines += ["SESSION ACTIVITY (one line per observation, oldest first):"] + lines += [f"- {line}" for line in obs_lines] + lines += [""] + if changed_files: + lines += ["FILES CHANGED:"] + lines += [f"- {f}" for f in changed_files[:50]] + lines += [""] + if git_stat: + lines += ["GIT STAT:", "```", git_stat, "```", ""] + + pages = store.list_pages() + pending = compile_mod._pending_page_names(store) + taken = [f"- {p.title}" for p in pages] + [f"- {n} [pending]" for n in sorted(pending)] + lines += ["TAKEN TOPICS (do NOT redraft any of these):"] + lines += taken or ["- (none)"] + lines += ["", *_rules_lines(max_pages)] + return "\n".join(lines), truncated + + +def _rules_lines(max_pages: int) -> list[str]: + """The shared clustering rules + JSON output contract for both the + buffer split and the re-narrate-from-record prompts.""" + return [ + "RULES", + f"- Cluster the activity into at most {max_pages} coherent TOPICS —", + " distinct threads of work in this session. Draft one page per topic.", + "- Each page needs a specific title (\"fixed the audit-log write race\",", + " not \"bug fixes\") and an 80-200 word markdown body summarizing that", + " thread of work.", + "- These are session records, NOT wiki topic pages: do NOT add", + " [claim: id] markers, and do NOT invent facts beyond the activity shown.", + "- Skip any topic already listed under TAKEN TOPICS.", + "", + "OUTPUT: print ONLY a JSON array, no code fences, no commentary.", + "Each element: {\"title\": str, \"body\": str}", + ] + + +def _skip(session_id: str, reason: str, *, proposal_id: str | None = None) -> dict[str, Any]: + return {"captured": 0, "summary_proposal_id": proposal_id, "summary_proposal_ids": [], + "mode": "skipped", "skipped": reason, "session_id": session_id, + "summarized": False, "proposal_id": proposal_id} + + +def _eligible_mechanical_proposal(store: KBStore, session_id: str) -> Any | None: + """The pending, not-yet-narrated session summary for `session_id`, if any. + + Only mechanical rollups (proposed by vouch-capture) are eligible; a + session-split proposal is already narrated. + """ + for prop in store.list_proposals(ProposalStatus.PENDING): + if prop.kind.value != "page": + continue + if str(prop.payload.get("type") or "") != capture.CAPTURE_PAGE_TYPE: + continue + if prop.session_id != session_id: + continue + if prop.proposed_by != capture.CAPTURE_ACTOR: + continue + return prop + return None + + +def build_renarrate_prompt(store: KBStore, body: str, *, title: str, max_pages: int) -> str: + """Prompt to narrate an already-filed mechanical session record into + topical pages. Source is the filed proposal's markdown body (the buffer is + gone by the time a filed summary is re-narrated).""" + lines = [ + "You are the session historian for this project's knowledge base. You are", + "given a mechanically-generated session record. Rewrite it into coherent,", + "narrated topical pages — one per distinct thread of work.", + "", + ] + if title: + lines += [f"SESSION RECORD TITLE: {title}", ""] + lines += ["SESSION RECORD (markdown):", body, ""] + pages = store.list_pages() + pending = compile_mod._pending_page_names(store) + taken = [f"- {p.title}" for p in pages] + [f"- {n} [pending]" for n in sorted(pending)] + lines += ["TAKEN TOPICS (do NOT redraft any of these):"] + lines += taken or ["- (none)"] + lines += ["", *_rules_lines(max_pages)] + return "\n".join(lines) + + +def _try_renarrate( + store: KBStore, session_id: str, *, split_cfg: SplitConfig +) -> dict[str, Any] | None: + """Narrate a filed mechanical summary with the LLM, superseding it. + + Returns None when no eligible mechanical proposal exists (caller falls + through to the below-min skip). On success files narrated page proposals + and rejects the mechanical one. On LLM failure leaves it intact. + """ + prop = _eligible_mechanical_proposal(store, session_id) + if prop is None: + return None + cmd = split_cfg.llm_cmd or compile_mod.load_config(store).llm_cmd + if not cmd: + return _skip(session_id, "not-configured", proposal_id=prop.id) + body = str(prop.payload.get("body") or "") + title = str(prop.payload.get("title") or "") + try: + prompt = build_renarrate_prompt(store, body, title=title, max_pages=split_cfg.max_pages) + raw = llm_draft.run_llm( + cmd, prompt, timeout_seconds=split_cfg.timeout_seconds, + label="capture.split.llm_cmd", + ) + drafts = llm_draft.parse_drafts(raw, noun="page") + ids, dropped = _file_drafts(store, session_id, drafts, split_cfg.max_pages) + except LLMDraftError as e: + logger.warning("session_split: renarrate failed for %s (%s)", session_id, e) + return _skip(session_id, "llm-failed", proposal_id=prop.id) + if not ids: + logger.warning("session_split: renarrate produced no valid drafts for %s", session_id) + return _skip(session_id, "llm-failed", proposal_id=prop.id) + reject( + store, prop.id, rejected_by=SPLIT_ACTOR, + reason="superseded by llm narrative summary", + ) + _audit_split(store, session_id, ids, dropped, 0, False) + return {"captured": 0, "summary_proposal_id": ids[0], "summary_proposal_ids": ids, + "mode": "renarrated", "dropped": dropped, "truncated": False, + "session_id": session_id, "summarized": True, "proposal_id": ids[0], + "superseded": prop.id} + + +def _file_drafts( + store: KBStore, + session_id: str, + drafts: list[dict[str, Any]], + max_pages: int, +) -> tuple[list[str], list[dict[str, Any]]]: + existing = store.list_pages() + taken = {p.title.strip().lower() for p in existing} + taken |= {p.id.strip().lower() for p in existing} + taken |= compile_mod._pending_page_names(store) + ids: list[str] = [] + dropped: list[dict[str, Any]] = [] + for i, draft in enumerate(drafts): + title = str(draft.get("title") or "").strip() + body = str(draft.get("body") or "").strip() + if not title: + dropped.append({"title": f"draft {i}", "reason": "draft has no title"}) + continue + if not body: + dropped.append({"title": title, "reason": "draft has no body"}) + continue + if len(ids) >= max_pages: + dropped.append({"title": title, "reason": f"over max_pages={max_pages}"}) + continue + if title.lower() in taken or _slugify(title) in taken: + dropped.append({"title": title, "reason": "title already exists or is pending"}) + continue + proposal = propose_page( + store, title=title, body=body, + page_type=capture.CAPTURE_PAGE_TYPE, # "session" — forced, ignore any LLM type + proposed_by=SPLIT_ACTOR, + tags=["session", "split"], + session_id=session_id, + metadata={"session_id": session_id}, + rationale=f"llm topical split of session {session_id}", + ) + ids.append(proposal.id) + taken.add(title.lower()) + taken.add(_slugify(title)) + return ids, dropped + + +def _audit_split( + store: KBStore, + session_id: str, + ids: list[str], + dropped: list[dict[str, Any]], + n_observations: int, + truncated: bool, +) -> None: + audit_mod.log_event( + store.kb_dir, event="session.split", actor=SPLIT_ACTOR, + object_ids=ids, + data={"proposed": len(ids), "dropped": len(dropped), + "observations": n_observations, "truncated": truncated}, + ) + + +def build_session_rows(store: KBStore) -> list[dict[str, Any]]: + """Assemble the session-review pipeline view (`kb.list_sessions`). + + Two stages, host-blind (reads the same buffers + proposals every adapter's + capture feeds): + + - "pending": a filed session-summary page proposal awaiting review. + `summarized=True` — it already has a summary (mechanical or LLM-split). + - "buffer": an open capture buffer that has not been summarized yet. + `summarized=False` — it still needs a summary. + + A session with a filed proposal is not also listed as a buffer (finalize + deletes the buffer, but guard against a race). Newest activity first. + """ + rows: list[dict[str, Any]] = [] + pending_session_ids: set[str] = set() + + for prop in store.list_proposals(ProposalStatus.PENDING): + if prop.kind.value != "page": + continue + if str(prop.payload.get("type") or "") != capture.CAPTURE_PAGE_TYPE: + continue + if prop.session_id: + pending_session_ids.add(prop.session_id) + # a mechanical rollup (vouch-capture) still needs an LLM narrative; + # only a session-split proposal counts as already summarized. + tags = prop.payload.get("tags") or [] + summarized = prop.proposed_by == SPLIT_ACTOR or "split" in tags + rows.append({ + "session_id": prop.session_id, + "stage": "pending", + "proposal_id": prop.id, + "kind": "page", + "title": prop.payload.get("title"), + "summarized": summarized, + "observations": None, + "last_activity": prop.proposed_at.isoformat(), + }) + + caps = capture.captures_dir(store) + if caps.exists(): + for path in sorted(caps.glob("*.jsonl")): + sid = path.stem + if sid in pending_session_ids: + continue + obs = capture._read_observations(path) + ts_vals = [float(o["ts"]) for o in obs if o.get("ts") is not None] + last = ( + datetime.fromtimestamp(max(ts_vals), tz=UTC).isoformat() + if ts_vals else None + ) + rows.append({ + "session_id": sid, + "stage": "buffer", + "proposal_id": None, + "kind": None, + "title": None, + "summarized": False, + "observations": len(obs), + "last_activity": last, + }) + + rows.sort(key=lambda r: r["last_activity"] or "", reverse=True) + return rows diff --git a/src/vouch/sessions.py b/src/vouch/sessions.py index d243646b..a0079366 100644 --- a/src/vouch/sessions.py +++ b/src/vouch/sessions.py @@ -149,11 +149,16 @@ def _build_summary_body(sess: Session, ids: list[str]) -> str: # through the review gate. Anything agent-controlled (sess.task, # sess.note, sess.agent) is omitted to keep the review-gate guarantee # intact for the Page artifact kind. See #76. + # crystallize does not end the session, so ended_at is often None here. + # A wall-clock fallback would make every retry write a different body, + # breaking crystallize idempotency (#256) and re-embedding needlessly, so + # emit a stable marker instead of datetime.now() for a still-open session. + ended = sess.ended_at.isoformat() if sess.ended_at else "(session still open)" lines = [ f"# Session {sess.id}", "", f"**Started:** {sess.started_at.isoformat()}", - f"**Ended:** {(sess.ended_at or datetime.now(UTC)).isoformat()}", + f"**Ended:** {ended}", "", "## Crystallized artifacts", "", diff --git a/src/vouch/skills.py b/src/vouch/skills.py new file mode 100644 index 00000000..c2590764 --- /dev/null +++ b/src/vouch/skills.py @@ -0,0 +1,239 @@ +"""Skill / slash-command discovery for agents. + +Lets an MCP-connected agent (Claude Code, Cursor, …) introspect what +skills and slash commands are available in the current environment +without having to read the filesystem itself. The agent calls +``kb.list_skills`` to see what's installed, then ``kb.get_skill`` to +pull the body of one it wants to run. + +Scanned locations, in priority order (later wins on name collision): + +1. ``/.claude/skills//SKILL.md`` — project-local skills +2. ``/.claude/commands/.md`` — project-local slash commands +3. ``~/.claude/skills//SKILL.md`` — user-global skills +4. ``~/.claude/commands/.md`` — user-global slash commands + +Project-local skills override user-global ones with the same name so a +KB can ship its own slash-command flavour that masks the user's default. + +A skill description is parsed from YAML frontmatter when present +(``description:`` field, gbrain-style), otherwise from the first +markdown paragraph after the first heading. Discovery silently skips +unreadable files — the catalogue is best-effort. + +Publishing the catalogue over MCP is gated by ``mcp.publish_skills`` in +``config.yaml`` (default ``true``). When flipped to ``false`` — +"company-brain" mode where the slash-command catalogue is itself +sensitive — ``list_skills`` returns an empty list and ``get_skill`` +raises :class:`SkillsDisabledError`, which the MCP / JSONL layer maps to +a ``permission_denied`` error. The flag is read fresh on every call, so +toggling it takes effect without restarting the server. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import yaml + +from .storage import KBStore + +# Matches a YAML frontmatter block at the very top of a file. +_FRONTMATTER_RE = re.compile(r"\A---\s*\n(.*?)\n---\s*\n?(.*)\Z", re.DOTALL) +# After stripping frontmatter, find the first markdown heading. +_FIRST_HEADING_RE = re.compile(r"^#+\s+(.*?)\s*$", re.MULTILINE) +DESCRIPTION_PREVIEW_CHARS = 280 + + +class SkillsDisabledError(RuntimeError): + """Raised when the skill catalogue is requested while it's gated off. + + Surfaced when ``mcp.publish_skills`` is ``false`` in ``config.yaml``. + The JSONL / MCP layer translates this into a ``permission_denied`` + error rather than a generic failure. + """ + + +@dataclass(frozen=True) +class SkillRecord: + name: str + description: str + scope: str # "project" | "user" + kind: str # "skill" | "command" + path: str # absolute path on disk + + +def _load_cfg(store: KBStore) -> dict[str, Any]: + """Read ``config.yaml`` defensively — a missing or malformed file is + treated as an empty config (default-on behaviour).""" + try: + loaded = yaml.safe_load((store.kb_dir / "config.yaml").read_text()) + except Exception: + return {} + return loaded if isinstance(loaded, dict) else {} + + +def publish_skills_enabled(store: KBStore) -> bool: + """Whether the skill catalogue may be published over MCP. + + Controlled by ``mcp.publish_skills`` in ``config.yaml``. Defaults to + ``True`` so existing KBs with no ``mcp:`` block keep the catalogue. + Only an explicit ``false`` turns the gate off — any other value (or a + missing key) is treated as enabled. + """ + mcp = _load_cfg(store).get("mcp") + if not isinstance(mcp, dict): + return True + return mcp.get("publish_skills", True) is not False + + +def _parse_frontmatter(text: str) -> tuple[dict[str, Any], str]: + """Split a markdown file into (frontmatter_dict, body).""" + m = _FRONTMATTER_RE.match(text) + if not m: + return {}, text + try: + meta = yaml.safe_load(m.group(1)) or {} + if not isinstance(meta, dict): + return {}, text + except yaml.YAMLError: + return {}, text + return meta, m.group(2) + + +def _derive_description(meta: dict[str, Any], body: str) -> str: + """Prefer explicit ``description:`` frontmatter; fall back to the first + paragraph of body text after the first heading.""" + desc = meta.get("description") + if isinstance(desc, str) and desc.strip(): + flat = " ".join(desc.strip().split()) + return flat[:DESCRIPTION_PREVIEW_CHARS] + + # Drop the first heading line, then take the first non-empty paragraph. + stripped = _FIRST_HEADING_RE.sub("", body, count=1).strip() + paragraphs = [p.strip() for p in stripped.split("\n\n") if p.strip()] + if not paragraphs: + return "" + flat = " ".join(paragraphs[0].split()) + if len(flat) <= DESCRIPTION_PREVIEW_CHARS: + return flat + return flat[: DESCRIPTION_PREVIEW_CHARS - 1] + "…" + + +def _read_skill_file( + path: Path, *, scope: str, kind: str, fallback_name: str, +) -> SkillRecord | None: + """Parse one skill file. Returns None if the file is unreadable.""" + try: + text = path.read_text(encoding="utf-8") + except OSError: + return None + meta, body = _parse_frontmatter(text) + name = meta.get("name") if isinstance(meta.get("name"), str) else None + name = (name or fallback_name).strip() + description = _derive_description(meta, body) + return SkillRecord( + name=name, + description=description, + scope=scope, + kind=kind, + path=str(path), + ) + + +def _scan_dir( + base: Path, *, scope: str, +) -> list[SkillRecord]: + """Walk one of the four catalogue roots and yield records.""" + records: list[SkillRecord] = [] + skills_root = base / "skills" + if skills_root.is_dir(): + for sub in sorted(skills_root.iterdir()): + skill_md = sub / "SKILL.md" + if skill_md.is_file(): + rec = _read_skill_file( + skill_md, scope=scope, kind="skill", fallback_name=sub.name, + ) + if rec is not None: + records.append(rec) + + commands_root = base / "commands" + if commands_root.is_dir(): + for md in sorted(commands_root.glob("*.md")): + rec = _read_skill_file( + md, scope=scope, kind="command", fallback_name=md.stem, + ) + if rec is not None: + records.append(rec) + + return records + + +def _catalogue(store: KBStore) -> dict[str, SkillRecord]: + """Build the merged catalogue. Project-local entries override user ones.""" + # User-global first, then project — project overwrites on collision. + user_base = Path.home() / ".claude" + project_base = store.root / ".claude" + + merged: dict[str, SkillRecord] = {} + for rec in _scan_dir(user_base, scope="user"): + merged[rec.name] = rec + for rec in _scan_dir(project_base, scope="project"): + merged[rec.name] = rec + return merged + + +def list_skills(store: KBStore) -> list[dict[str, Any]]: + """Catalogue every discoverable skill / slash command. + + Returns one row per skill: ``{name, description, scope, kind, path}``. + Sorted by name for stable output. Returns an empty list when + ``mcp.publish_skills`` is ``false`` — the catalogue is hidden without + erroring so a polling client just sees "nothing installed". + """ + if not publish_skills_enabled(store): + return [] + cat = _catalogue(store) + return [ + { + "name": r.name, + "description": r.description, + "scope": r.scope, + "kind": r.kind, + "path": r.path, + } + for r in sorted(cat.values(), key=lambda r: r.name) + ] + + +def get_skill(store: KBStore, name: str) -> dict[str, Any]: + """Return the full markdown body of a named skill. + + Raises :class:`SkillsDisabledError` when ``mcp.publish_skills`` is + ``false`` (mapped to ``permission_denied`` by the transport), and + ``KeyError`` if the name isn't in the catalogue (mapped to a clean + ``not_found`` error). + """ + if not publish_skills_enabled(store): + raise SkillsDisabledError( + "skill catalogue is disabled (mcp.publish_skills: false)" + ) + cat = _catalogue(store) + rec = cat.get(name) + if rec is None: + raise KeyError(f"unknown skill: {name!r}") + try: + body = Path(rec.path).read_text(encoding="utf-8") + except OSError as e: + raise KeyError(f"could not read skill {name!r}: {e}") from e + return { + "name": rec.name, + "description": rec.description, + "scope": rec.scope, + "kind": rec.kind, + "path": rec.path, + "body": body, + } diff --git a/src/vouch/stats.py b/src/vouch/stats.py index 5e5ca35c..db30754f 100644 --- a/src/vouch/stats.py +++ b/src/vouch/stats.py @@ -7,14 +7,19 @@ from __future__ import annotations import statistics +from collections.abc import Callable from datetime import UTC, datetime, timedelta -from typing import Any +from typing import TYPE_CHECKING, Any +from zoneinfo import ZoneInfo, ZoneInfoNotFoundError from . import audit, health from .models import Proposal, ProposalStatus from .proposals import EXPIRE_REASON from .storage import KBStore, _yaml_load +if TYPE_CHECKING: + from .scoping import ViewerContext + def _utc(dt: datetime) -> datetime: if dt.tzinfo is None: @@ -186,6 +191,103 @@ def bump(agent: str, bucket: str) -> None: } +# Real-world UTC offsets span -12:00..+14:00; clamp so a bad client can't +# shift events into arbitrary buckets. +_MAX_TZ_OFFSET_MINUTES = 14 * 60 + + +def _is_proposal_create(event: str) -> bool: + return event.startswith("proposal.") and event.endswith(".create") + + +def _local_clock(tz: str | None, offset_minutes: int) -> Callable[[datetime], datetime]: + """Viewer-local conversion: an IANA zone when resolvable (DST-correct), + otherwise the fixed offset.""" + if tz: + try: + zone = ZoneInfo(tz) + except (ZoneInfoNotFoundError, ValueError): + pass + else: + return lambda dt: _utc(dt).astimezone(zone) + shift = timedelta(minutes=offset_minutes) + return lambda dt: _utc(dt) + shift + + +def collect_activity( + store: KBStore, + *, + days: int = 365, + tz_offset_minutes: int = 0, + tz: str | None = None, + viewer: ViewerContext | None = None, +) -> dict[str, Any]: + """Bucket audit events for the console dashboard — `kb.activity`. + + One pass over the audit log: per-day totals (with proposal/decision + breakdowns), an hour-of-week matrix, and actor/event histograms. + ``days=0`` means all-time; otherwise the window is the last ``days`` + viewer-local calendar days including today, so the oldest day in a + dashboard heatmap is never partially counted. ``tz`` (IANA name) wins + over ``tz_offset_minutes`` for local bucketing. ``viewer`` applies the + same scope filtering as `kb.audit`. + """ + if days < 0: + raise ValueError("days must be >= 0") + window = None if days == 0 else days + offset = max(-_MAX_TZ_OFFSET_MINUTES, min(_MAX_TZ_OFFSET_MINUTES, tz_offset_minutes)) + to_local = _local_clock(tz, offset) + cutoff_day = ( + None + if window is None + else to_local(datetime.now(UTC)).date() - timedelta(days=window - 1) + ) + + by_day: dict[str, dict[str, int]] = {} + # by_hour[weekday][hour], Monday = 0 — matches datetime.weekday(). + by_hour = [[0] * 24 for _ in range(7)] + by_actor: dict[str, int] = {} + by_event: dict[str, int] = {} + total = 0 + + for ev in audit.read_events(store.kb_dir, store=store, viewer=viewer): + local = to_local(ev.created_at) + if cutoff_day is not None and local.date() < cutoff_day: + continue + day = by_day.setdefault( + local.date().isoformat(), + {"total": 0, "proposals": 0, "decisions": 0}, + ) + day["total"] += 1 + if _is_proposal_create(ev.event): + day["proposals"] += 1 + elif _audit_decision_kind(ev.event) is not None: + day["decisions"] += 1 + by_hour[local.weekday()][local.hour] += 1 + by_actor[ev.actor] = by_actor.get(ev.actor, 0) + 1 + by_event[ev.event] = by_event.get(ev.event, 0) + 1 + total += 1 + + days_seen = sorted(by_day) + return { + "generated_at": datetime.now(UTC).isoformat(), + "window_days": window, + "tz_offset_minutes": offset, + "viewer": { + "project": viewer.project if viewer else None, + "agent": viewer.agent if viewer else None, + }, + "total_events": total, + "active_days": len(by_day), + "first_event_day": days_seen[0] if days_seen else None, + "last_event_day": days_seen[-1] if days_seen else None, + "by_day": {d: by_day[d] for d in days_seen}, + "by_hour": by_hour, + "by_actor": dict(sorted(by_actor.items(), key=lambda kv: (-kv[1], kv[0]))), + "by_event": dict(sorted(by_event.items(), key=lambda kv: (-kv[1], kv[0]))), + } + + def collect_stats(store: KBStore, *, since_days: int | None = 30) -> dict[str, Any]: """Aggregate observability metrics for the KB at ``store``.""" return { diff --git a/src/vouch/storage.py b/src/vouch/storage.py index 3a057db3..b2214d7b 100644 --- a/src/vouch/storage.py +++ b/src/vouch/storage.py @@ -81,9 +81,19 @@ def _starter_config() -> dict[str, Any]: "expire_pending_after_days": 90, }, "capture": { - # auto-capture claude code sessions into pending summaries. + # auto-capture agent sessions into pending summaries. "enabled": True, "min_observations": 3, + "split": { + # llm topical split for large sessions; llm_cmd falls back to + # compile.llm_cmd when null. see session_split.py. + "enabled": True, + "llm_cmd": None, + "threshold_observations": 40, + "max_pages": 6, + "timeout_seconds": 180, + "max_input_chars": 60000, + }, }, "recall": { # inject a digest of all approved knowledge at session start. @@ -103,6 +113,12 @@ def _starter_config() -> dict[str, Any]: "human review via vouch pending/show/approve", ], }, + "mcp": { + # Publish the slash-command / SKILL.md catalogue over MCP via + # kb.list_skills / kb.get_skill. Flip to false for "company-brain" + # mode where the catalogue itself is sensitive. + "publish_skills": True, + }, # Extra page kinds beyond the built-in PageType enum. Each maps a kind # name to {required_fields, frontmatter_schema, required_citations, # extends}. See `vouch schema list` / docs for the shape. (issue #234) @@ -193,6 +209,15 @@ def _deserialize_page(text: str) -> Page: return Page(body=body, **meta) +def _load_page_or_skip(path: Path) -> Page | None: + """Parse one page file; skip corrupt/unreadable files like ``_load_or_skip``.""" + try: + return _deserialize_page(path.read_text(encoding="utf-8")) + except (yaml.YAMLError, ValidationError, UnicodeDecodeError, OSError, ValueError) as e: + _log.warning("skipping unreadable page %s: %s", path.name, e) + return None + + class KBStore: """File-backed CRUD layer. Pure I/O — no business logic.""" @@ -550,8 +575,9 @@ def list_pages(self) -> list[Page]: if not pdir.is_dir(): return [] return [ - _deserialize_page(p.read_text(encoding="utf-8")) + pg for p in sorted(pdir.glob("*.md")) + if (pg := _load_page_or_skip(p)) is not None ] # --- entities ---------------------------------------------------------- @@ -654,6 +680,34 @@ def put_relation_idempotent(self, rel: Relation) -> Relation: ) return rel + def delete_claim(self, claim_id: str) -> None: + """Remove a claim file. Pure I/O; ref checks live in `proposals`.""" + path = self._claim_path(claim_id) + if not path.exists(): + raise ArtifactNotFoundError(f"claim {claim_id}") + path.unlink() + + def delete_page(self, page_id: str) -> None: + """Remove a page file. Pure I/O; ref checks live in `proposals`.""" + path = self._page_path(page_id) + if not path.exists(): + raise ArtifactNotFoundError(f"page {page_id}") + path.unlink() + + def delete_entity(self, entity_id: str) -> None: + """Remove an entity file. Pure I/O; ref checks live in `proposals`.""" + path = self._entity_path(entity_id) + if not path.exists(): + raise ArtifactNotFoundError(f"entity {entity_id}") + path.unlink() + + def delete_relation(self, relation_id: str) -> None: + """Remove a relation file. Pure I/O; ref checks live in `proposals`.""" + path = self._relation_path(relation_id) + if not path.exists(): + raise ArtifactNotFoundError(f"relation {relation_id}") + path.unlink() + def get_relation(self, rid: str) -> Relation: p = self._relation_path(rid) if not p.exists(): diff --git a/src/vouch/synthesize.py b/src/vouch/synthesize.py index a45eee73..20577b85 100644 --- a/src/vouch/synthesize.py +++ b/src/vouch/synthesize.py @@ -7,17 +7,25 @@ no claim for in an explicit `gaps` block, and grades its own confidence from the lifecycle status of the claims it cited. -The synthesis is deterministic in v1 — there is no LLM in the loop. The -`llm` flag is reserved so the wire shape is stable when an opt-in generative -backend lands; passing `llm=True` raises rather than silently degrading. +The default synthesis is deterministic — no LLM in the loop. `llm=True` +opts into the generative backend: the deployment-configured LLM command +(``compile.llm_cmd``, the same one the wiki compiler uses) drafts prose +grounded in retrieved *pages* and approved claims. The division of labor +stays llm-wiki's: the LLM writes, code verifies — every ``[id]`` citation +in the draft must resolve to a source that was actually offered, or it is +stripped; an answer left with no verifiable citation is returned empty +rather than uncited. Requesting `llm=True` without a configured command +raises rather than silently degrading. """ from __future__ import annotations +import re from typing import Any, Literal +from . import llm_draft from .context import build_context_pack -from .models import Claim, ClaimStatus +from .models import Claim, ClaimStatus, Page, PageStatus from .storage import ArtifactNotFoundError, KBStore Confidence = Literal["high", "medium", "low"] @@ -71,6 +79,183 @@ def _confidence(statuses: list[ClaimStatus]) -> Confidence: return "medium" +# LLM-backend grounding budgets. Pages are the product (llm-wiki): they get +# the deep budget; claims ground the sentences pages haven't compiled yet. +_LLM_PAGE_LIMIT = 6 +_LLM_CLAIM_LIMIT = 12 +_LLM_PAGE_CHARS = 4000 + +# Bracketed tokens in the draft. Anything bracketed that is not an offered +# source id is stripped — the model must not mint provenance. +_MARKER_RE = re.compile(r"\s*\[([^\[\]]+)\]") + + +def _llm_grounding( + store: KBStore, query: str, depth: int +) -> tuple[list[Page], list[Claim]]: + """Retrieve the pages and approved claims the LLM may cite. + + Retrieval-first via the context pack; when the index surfaces no page for + the query, fall back to the most recently updated pages so the chat still + checks the wiki on small or sparsely-indexed KBs. + """ + pack = build_context_pack(store, query=query, limit=max(depth * 4, 12)) + items = pack["items"] if isinstance(pack, dict) else pack.items + page_ids: list[str] = [] + claim_ids: list[str] = [] + for item in items: + kind = item["type"] if isinstance(item, dict) else item.type + iid = item["id"] if isinstance(item, dict) else item.id + bucket = page_ids if kind == "page" else claim_ids if kind == "claim" else None + if bucket is not None and iid not in bucket: + bucket.append(iid) + + pages: list[Page] = [] + for pid in page_ids: + if len(pages) >= _LLM_PAGE_LIMIT: + break + try: + page = store.get_page(pid) + except ArtifactNotFoundError: + continue + if page.status != PageStatus.ARCHIVED: + pages.append(page) + if not pages: + live = [p for p in store.list_pages() if p.status != PageStatus.ARCHIVED] + live.sort(key=lambda p: p.updated_at, reverse=True) + pages = live[:_LLM_PAGE_LIMIT] + + claims: list[Claim] = [] + for cid in claim_ids: + if len(claims) >= _LLM_CLAIM_LIMIT: + break + try: + claims.append(store.get_claim(cid)) + except ArtifactNotFoundError: + continue + return pages, claims + + +def _llm_prompt( + query: str, pages: list[Page], claims: list[Claim], max_chars: int +) -> str: + lines = [ + "You answer a question from a review-gated knowledge base.", + "Use ONLY the sources below — never outside knowledge.", + "After every sentence, cite the id of the supporting source in square " + "brackets, e.g. [some-id]. Only ids listed below count as citations.", + "If part of the question is not covered by the sources, do not guess — " + "name that topic in gaps instead.", + f"Keep the answer under {max_chars} characters.", + "Output exactly one JSON object and nothing else, no code fence:", + '{"answer": "…", "gaps": ["uncovered topic", "…"]}', + "", + f"Question: {query}", + "", + "Pages:", + ] + for page in pages: + body = page.body.strip() + if len(body) > _LLM_PAGE_CHARS: + body = body[:_LLM_PAGE_CHARS] + " …" + lines.append(f"[{page.id}] {page.title}\n{body}\n") + if not pages: + lines.append("(none)") + lines.append("Claims:") + for claim in claims: + lines.append(f"[{claim.id}] ({claim.status.value}) {claim.text}") + if not claims: + lines.append("(none)") + return "\n".join(lines) + + +def _llm_synthesize( + store: KBStore, *, query: str, depth: int, max_chars: int +) -> dict[str, Any]: + # The LLM command is deployment config shared with the wiki compiler — + # one knob (compile.llm_cmd) turns on every generative feature. + from .compile import load_config + + cfg = load_config(store) + if not cfg.llm_cmd: + raise ValueError( + "llm synthesis is not configured — set compile.llm_cmd in " + '.vouch/config.yaml, e.g.\ncompile:\n llm_cmd: "claude -p --model sonnet"' + ) + + pages, claims = _llm_grounding(store, query, depth) + if not pages and not claims: + return { + "query": query, + "answer": "", + "claims": [], + "pages": [], + "gaps": _salient_terms(query), + "_meta": { + "synthesis_confidence": _confidence([]), + "synthesis_backend": "llm", + }, + } + + prompt = _llm_prompt(query, pages, claims, max_chars) + try: + raw = llm_draft.run_llm( + cfg.llm_cmd, prompt, + timeout_seconds=cfg.timeout_seconds, label="compile.llm_cmd", + ) + data = llm_draft.parse_object(raw, noun="synthesis") + except llm_draft.LLMDraftError as e: + raise ValueError(str(e)) from e + + answer = str(data.get("answer") or "").strip() + gaps = [str(g) for g in data.get("gaps") or [] if str(g).strip()] + + # Code verifies what the model drafted: strip any bracketed token that is + # not an offered source id, then truncate to budget on a citation boundary + # so no dangling half-citation survives. + offered = {p.id for p in pages} | {c.id for c in claims} + dropped: list[str] = [] + + def _keep(m: re.Match[str]) -> str: + if m.group(1) in offered: + return m.group(0) + dropped.append(m.group(1)) + return "" + + answer = _MARKER_RE.sub(_keep, answer) + answer = re.sub(r" +([.,;:])", r"\1", re.sub(r" {2,}", " ", answer)).strip() + if len(answer) > max_chars: + cut = answer.rfind("]", 0, max_chars) + answer = answer[: cut + 1] if cut > 0 else answer[:max_chars] + + cited = [m.group(1) for m in _MARKER_RE.finditer(answer)] + cited = list(dict.fromkeys(cited)) + if not cited: + # an uncited draft is a guess — the KB stays silent instead + answer = "" + gaps = gaps or _salient_terms(query) + + claim_by_id = {c.id: c for c in claims} + cited_claims = [cid for cid in cited if cid in claim_by_id] + cited_pages = [cid for cid in cited if cid not in claim_by_id] + meta: dict[str, Any] = { + "synthesis_confidence": _confidence( + [claim_by_id[cid].status for cid in cited_claims] + ), + "synthesis_backend": "llm", + } + if dropped: + meta["dropped_citations"] = dropped + return { + "query": query, + "answer": answer, + "claims": cited_claims, + "pages": cited_pages, + "gaps": gaps, + "_meta": meta, + } + + def synthesize( store: KBStore, *, @@ -79,16 +264,18 @@ def synthesize( max_chars: int = 4000, llm: bool = False, ) -> dict[str, Any]: - """Answer `query` from approved claims only, with inline citations. + """Answer `query` from the review-gated KB, with inline citations. Returns a dict with `query`, `answer` (citation-bearing prose, possibly - empty), `claims` (the cited claim ids), `gaps` (query topics no approved - claim covered) and `_meta.synthesis_confidence`. + empty), `claims` (the cited claim ids), `pages` (cited page ids — always + empty on the deterministic path), `gaps` (query topics no source covered) + and `_meta.synthesis_confidence`. With `llm=True` the answer is drafted + by the deployment-configured LLM grounded in pages and approved claims; + every citation is still verified mechanically. """ if llm: - raise ValueError( - "llm synthesis backend not configured; " - "deterministic synthesis is the default" + return _llm_synthesize( + store, query=query, depth=depth, max_chars=max_chars, ) pack = build_context_pack(store, query=query, limit=depth) @@ -135,6 +322,10 @@ def synthesize( "query": query, "answer": answer, "claims": cited, + "pages": [], "gaps": gaps, - "_meta": {"synthesis_confidence": _confidence(statuses)}, + "_meta": { + "synthesis_confidence": _confidence(statuses), + "synthesis_backend": "deterministic", + }, } diff --git a/src/vouch/transcript.py b/src/vouch/transcript.py new file mode 100644 index 00000000..5ca10d34 --- /dev/null +++ b/src/vouch/transcript.py @@ -0,0 +1,406 @@ +"""Locate and parse raw agent session transcripts on demand. + +Given a captured session id, find the raw JSONL the agent wrote (Claude Code +under ``~/.claude/projects``, Codex rollouts under ``$CODEX_HOME/sessions``) +and normalize it into a block schema the vouch console renders. Read-only: +never writes to the KB. When the raw file is gone we degrade to vouch's +compact capture observations instead. +""" + +from __future__ import annotations + +import json +import os +import re +from pathlib import Path +from typing import TYPE_CHECKING, Any + +from .capture import _read_observations, buffer_path + +if TYPE_CHECKING: + from .storage import KBStore + +MAX_FILE_BYTES = 25 * 1024 * 1024 +MAX_MESSAGES = 2000 + +# Session ids are UUID-shaped; reject anything else so a hostile id can't +# widen a glob or traverse out of the projects tree. +_VALID_ID = re.compile(r"^[0-9a-fA-F-]{8,64}$") + + +def _claude_projects_root() -> Path: + env = os.environ.get("VOUCH_CLAUDE_PROJECTS_DIR") + return Path(env) if env else Path.home() / ".claude" / "projects" + + +def find_claude_file(session_id: str) -> Path | None: + """The raw Claude Code JSONL for ``session_id``, or None. + + Claude names each session file ``.jsonl`` under a per-cwd project + dir; subagent transcripts live under ``/subagents/**``. The file + stem is the id, so a literal name match (no id interpolation into a glob) + locates it. + """ + if not _VALID_ID.match(session_id): + return None + root = _claude_projects_root() + if not root.is_dir(): + return None + name = f"{session_id}.jsonl" + for project in root.iterdir(): + if not project.is_dir(): + continue + top = project / name + if top.is_file(): + return top + for candidate in root.glob(f"*/*/subagents/**/{name}"): + if candidate.is_file(): + return candidate + return None + + +def _norm_tokens(usage: dict[str, Any]) -> dict[str, int]: + def i(key: str) -> int: + v = usage.get(key) + return int(v) if isinstance(v, (int, float)) else 0 + + return { + "input": i("input_tokens"), + "output": i("output_tokens"), + "cache_read": i("cache_read_input_tokens"), + "cache_creation": i("cache_creation_input_tokens"), + } + + +def _result_text(content: Any) -> str: + """tool_result.content is a string, or a list of {type:text,text} parts.""" + if isinstance(content, str): + return content + if isinstance(content, list): + parts = [ + str(p.get("text", "")) + for p in content + if isinstance(p, dict) and p.get("type") == "text" + ] + if parts: + return "\n".join(parts) + return json.dumps(content, ensure_ascii=False) + if content is None: + return "" + return json.dumps(content, ensure_ascii=False) + + +def parse_claude_transcript(path: Path, *, max_messages: int = 2000) -> dict[str, Any]: + """Parse a Claude Code JSONL into the normalized transcript schema. + + Single forward pass: assistant content blocks that share a + ``message.id`` merge into one logical message; a later ``tool_result`` + (in a user entry) is paired into the matching ``tool_use`` block by id and + its user entry is not emitted as a standalone message. + """ + messages: list[dict[str, Any]] = [] + tool_by_id: dict[str, dict[str, Any]] = {} + session: dict[str, Any] = { + "id": path.stem, "agent": "claude", "cwd": None, "git_branch": None, + "title": None, "started_at": None, "ended_at": None, "model": None, + "tokens": {"input": 0, "output": 0, "cache_read": 0, "cache_creation": 0}, + } + truncated = False + current: dict[str, Any] | None = None + + def flush() -> None: + nonlocal current + if current is not None and current["blocks"]: + messages.append(current) + current = None + + with path.open(encoding="utf-8") as fh: + for raw in fh: + raw = raw.strip() + if not raw: + continue + try: + obj = json.loads(raw) + except json.JSONDecodeError: + continue + if not isinstance(obj, dict): + continue + if session["cwd"] is None and isinstance(obj.get("cwd"), str): + session["cwd"] = obj["cwd"] + if session["git_branch"] is None and isinstance(obj.get("gitBranch"), str): + session["git_branch"] = obj["gitBranch"] + ts = obj.get("timestamp") + if isinstance(ts, str): + if session["started_at"] is None: + session["started_at"] = ts + session["ended_at"] = ts + t = obj.get("type") + if t == "ai-title" and isinstance(obj.get("aiTitle"), str): + session["title"] = obj["aiTitle"] + continue + if t not in ("user", "assistant"): + continue + if len(messages) >= max_messages: + truncated = True + break + msg = obj.get("message") + if not isinstance(msg, dict): + continue + content = msg.get("content") + + if t == "assistant": + mid = msg.get("id") if isinstance(msg.get("id"), str) else None + if current is None or current.get("id") != mid: + flush() + raw_usage = msg.get("usage") + usage: dict[str, Any] = raw_usage if isinstance(raw_usage, dict) else {} + model = msg.get("model") if isinstance(msg.get("model"), str) else None + if model and session["model"] is None: + session["model"] = model + current = { + "role": "assistant", "id": mid, "model": model, + "timestamp": ts if isinstance(ts, str) else None, + "tokens": _norm_tokens(usage), "blocks": [], + } + tok = current["tokens"] + for k in session["tokens"]: + session["tokens"][k] += tok[k] + parts = content if isinstance(content, list) else [] + for part in parts: + if not isinstance(part, dict): + continue + ptype = part.get("type") + if ptype == "thinking": + text = str(part.get("thinking", "")).strip() + if text: + current["blocks"].append({"type": "thinking", "text": text}) + elif ptype == "text": + text = str(part.get("text", "")).strip() + if text: + current["blocks"].append({"type": "text", "text": text}) + elif ptype == "tool_use": + tid = part.get("id") + raw_input = part.get("input") + block: dict[str, Any] = { + "type": "tool_use", "id": tid, + "name": str(part.get("name", "")), + "input": raw_input if isinstance(raw_input, dict) else {}, + "result": None, + } + current["blocks"].append(block) + if isinstance(tid, str): + tool_by_id[tid] = block + continue + + # user entry + flush() + if isinstance(content, str): + text = content.strip() + if text: + messages.append({ + "role": "user", "id": None, "model": None, + "timestamp": ts if isinstance(ts, str) else None, + "tokens": None, "blocks": [{"type": "text", "text": text}], + }) + continue + parts = content if isinstance(content, list) else [] + user_blocks: list[dict[str, Any]] = [] + agent_id = None + tur = obj.get("toolUseResult") + if isinstance(tur, dict) and isinstance(tur.get("agentId"), str): + agent_id = tur["agentId"] + for part in parts: + if not isinstance(part, dict): + continue + if part.get("type") == "tool_result": + tid = part.get("tool_use_id") + paired = tool_by_id.get(tid) if isinstance(tid, str) else None + if paired is not None: + paired["result"] = { + "content": _result_text(part.get("content")), + "is_error": bool(part.get("is_error", False)), + "subagent_session_id": agent_id, + } + elif part.get("type") == "text": + text = str(part.get("text", "")).strip() + if text: + user_blocks.append({"type": "text", "text": text}) + if user_blocks: + messages.append({ + "role": "user", "id": None, "model": None, + "timestamp": ts if isinstance(ts, str) else None, + "tokens": None, "blocks": user_blocks, + }) + flush() + return {"session": session, "messages": messages, "truncated": truncated} + + +def _codex_message_text(content: Any) -> str: + """Join a codex message's content parts (input_text / output_text / text).""" + if isinstance(content, str): + return content.strip() + if not isinstance(content, list): + return "" + parts = [ + str(p.get("text", "")) + for p in content + if isinstance(p, dict) and p.get("type") in ("input_text", "output_text", "text") + ] + return "\n".join(x for x in parts if x).strip() + + +def parse_codex_transcript(path: Path, *, max_messages: int = 2000) -> dict[str, Any]: + """Parse a Codex rollout JSONL into the normalized transcript schema. + + The canonical conversation is the sequence of ``response_item`` records: + ``message`` (role user/assistant; developer/system boilerplate skipped), + ``function_call`` / ``custom_tool_call`` and their ``*_output`` pairs, and + ``reasoning`` (encrypted, so dropped). Assistant activity between user + messages groups into one assistant message; an output pairs into its call + by ``call_id``. ``session_meta`` supplies cwd / branch / timestamps. + """ + session: dict[str, Any] = { + "id": path.stem, "agent": "codex", "cwd": None, "git_branch": None, + "title": None, "started_at": None, "ended_at": None, "model": None, + "tokens": {"input": 0, "output": 0, "cache_read": 0, "cache_creation": 0}, + } + messages: list[dict[str, Any]] = [] + tool_by_call: dict[str, dict[str, Any]] = {} + truncated = False + current: dict[str, Any] | None = None + + def new_assistant() -> dict[str, Any]: + return { + "role": "assistant", "id": None, "model": None, + "timestamp": None, "tokens": None, "blocks": [], + } + + def flush() -> None: + nonlocal current + if current is not None and current["blocks"]: + messages.append(current) + current = None + + with path.open(encoding="utf-8") as fh: + for raw in fh: + raw = raw.strip() + if not raw: + continue + try: + rec = json.loads(raw) + except json.JSONDecodeError: + continue + if not isinstance(rec, dict): + continue + payload = rec.get("payload") + if not isinstance(payload, dict): + continue + rtype = rec.get("type") + + if rtype == "session_meta": + sid = payload.get("id") or payload.get("session_id") + if isinstance(sid, str) and sid.strip(): + session["id"] = sid.strip() + if isinstance(payload.get("cwd"), str): + session["cwd"] = payload["cwd"] + ts = payload.get("timestamp") + if isinstance(ts, str): + session["started_at"] = ts + session["ended_at"] = ts + git = payload.get("git") + if isinstance(git, dict) and isinstance(git.get("branch"), str): + session["git_branch"] = git["branch"] + continue + + if rtype != "response_item": + continue + if len(messages) >= max_messages: + truncated = True + break + ptype = payload.get("type") + + if ptype == "message": + role = payload.get("role") + text = _codex_message_text(payload.get("content")) + if role == "user": + flush() + if text: + messages.append({ + "role": "user", "id": None, "model": None, + "timestamp": None, "tokens": None, + "blocks": [{"type": "text", "text": text}], + }) + elif role == "assistant": + if current is None: + current = new_assistant() + if text: + current["blocks"].append({"type": "text", "text": text}) + # developer / system messages are instruction boilerplate: skip. + elif ptype in ("function_call", "custom_tool_call"): + if current is None: + current = new_assistant() + cid = payload.get("call_id") + if ptype == "function_call": + tool_input: dict[str, Any] = {"arguments": payload.get("arguments")} + else: + tool_input = {"input": payload.get("input")} + block: dict[str, Any] = { + "type": "tool_use", "id": cid, + "name": str(payload.get("name", "")), + "input": tool_input, "result": None, + } + current["blocks"].append(block) + if isinstance(cid, str): + tool_by_call[cid] = block + elif ptype in ("function_call_output", "custom_tool_call_output"): + cid = payload.get("call_id") + paired = tool_by_call.get(cid) if isinstance(cid, str) else None + if paired is not None: + paired["result"] = { + "content": str(payload.get("output", "")), + "is_error": False, "subagent_session_id": None, + } + flush() + return {"session": session, "messages": messages, "truncated": truncated} + + +def _degraded(store: KBStore, session_id: str, reason: str) -> dict[str, Any]: + obs = _read_observations(buffer_path(store, session_id)) + return {"available": False, "reason": reason, "observations": obs} + + +def load_transcript( + store: KBStore, session_id: str, *, agent: str | None = None +) -> dict[str, Any]: + """Locate + parse the raw transcript for ``session_id``. + + ``agent`` restricts the search ("claude" | "codex"); when None both are + tried. Returns the normalized schema on success, or a degraded result + (compact capture observations) when the raw file is missing/too large. + """ + path: Path | None = None + source_agent = "" + if agent in (None, "claude"): + path = find_claude_file(session_id) + if path is not None: + source_agent = "claude" + if path is None and agent in (None, "codex"): + from . import codex_rollout + + path = codex_rollout.find_rollout_by_session_id(session_id) + if path is not None: + source_agent = "codex" + if path is None: + return _degraded(store, session_id, f"raw transcript not found for session {session_id}") + try: + size = path.stat().st_size + except OSError as e: + return _degraded(store, session_id, f"cannot read transcript: {e}") + if size > MAX_FILE_BYTES: + return _degraded(store, session_id, f"transcript too large to render ({size} bytes)") + + if source_agent == "claude": + parsed = parse_claude_transcript(path, max_messages=MAX_MESSAGES) + else: + parsed = parse_codex_transcript(path, max_messages=MAX_MESSAGES) + return {"available": True, "source": {"agent": source_agent, "path": str(path)}, **parsed} diff --git a/src/vouch/trust.py b/src/vouch/trust.py index 3bda4e8a..8d764d50 100644 --- a/src/vouch/trust.py +++ b/src/vouch/trust.py @@ -31,6 +31,7 @@ "kb.capabilities", "kb.status", "kb.stats", + "kb.activity", "kb.search", "kb.context", "kb.read_page", diff --git a/src/vouch/web/__init__.py b/src/vouch/web/__init__.py index e6c89ab3..ece1cfa5 100644 --- a/src/vouch/web/__init__.py +++ b/src/vouch/web/__init__.py @@ -32,6 +32,26 @@ def _require_web_extra() -> None: ) +def _require_console_deps() -> None: + """Fail with a clean message if the console's serve deps aren't installed. + + The React console needs only starlette (the app) + uvicorn (the server) — + a subset of the [web] extra; jinja2 is not required. + """ + missing: list[str] = [] + for name in ("starlette", "uvicorn"): + try: + __import__(name) + except ImportError: + missing.append(name) + if missing: + raise ImportError( + "vouch console needs the [web] extra. " + "Install with: pip install 'vouch-kb[web]' " + f"(missing: {', '.join(missing)})" + ) + + def create_app( # type: ignore[no-untyped-def] kb_root: str | None = None, *, diff --git a/src/vouch/web/console.py b/src/vouch/web/console.py new file mode 100644 index 00000000..7b5fa8a9 --- /dev/null +++ b/src/vouch/web/console.py @@ -0,0 +1,174 @@ +"""Serve the vendored React review console (the `webapp/` SPA) from vouch. + +The console is a static single-page app. It cannot call vouch cross-origin — +vouch deliberately sends no CORS headers — so it reaches every backend through +a same-origin ``/proxy/*`` bridge, passing the real endpoint in an +``X-Vouch-Target`` header. In dev that bridge is a vite plugin +(`webapp/plugins/vouch-proxy.ts`); this module reimplements it in Python so a +single ``pip install 'vouch-kb[web]'`` can serve the built SPA with no node. + +This is a *viewport*: the bridge only forwards bytes to a `vouch serve +--transport http` backend, which remains the sole path to the review gate. +""" + +from __future__ import annotations + +import urllib.error +import urllib.parse +import urllib.request +from pathlib import Path + +from starlette.applications import Starlette +from starlette.concurrency import run_in_threadpool +from starlette.requests import Request +from starlette.responses import FileResponse, JSONResponse, Response +from starlette.routing import Route + +_MODULE_DIR = Path(__file__).resolve().parent + +# Loopback peers a same-origin browser client can present. The bridge is +# refused to anything else unless the operator explicitly opts in — a +# third-party page must not be able to drive a local reviewer's backends. +_LOOPBACK = frozenset({"127.0.0.1", "::1", "::ffff:127.0.0.1"}) + +# Methods the SPA actually uses are GET (health/capabilities) and POST (rpc); +# accept the common verbs so the bridge stays transparent. +_PROXY_METHODS = ["GET", "POST", "PUT", "PATCH", "DELETE", "OPTIONS", "HEAD"] + +# A generous ceiling: an rpc that runs the compile/summary LLM is synchronous. +_PROXY_TIMEOUT = 300.0 + + +class ConsoleError(RuntimeError): + """The console cannot be served (no built SPA, or a missing dependency).""" + + +def _default_repo_dist() -> Path: + """`webapp/dist` relative to a source checkout (…/src/vouch/web → repo).""" + return _MODULE_DIR.parents[2] / "webapp" / "dist" + + +def resolve_console_dir( + *, packaged: Path | None = None, repo_dist: Path | None = None +) -> Path | None: + """Locate the built console assets, or ``None`` if none are built. + + Prefers the copy bundled inside the wheel (``vouch/web/console``); falls + back to ``webapp/dist`` in a source checkout. Mirrors how + ``install_adapter`` prefers the repo tree over the packaged copy. + """ + packaged = packaged if packaged is not None else _MODULE_DIR / "console" + if (packaged / "index.html").is_file(): + return packaged + repo_dist = repo_dist if repo_dist is not None else _default_repo_dist() + if (repo_dist / "index.html").is_file(): + return repo_dist + return None + + +def _err(status: int, code: str, message: str) -> JSONResponse: + """The vouch-native error envelope the SPA already understands.""" + return JSONResponse( + {"ok": False, "error": {"code": code, "message": message}}, status_code=status + ) + + +def build_console_app(console_dir: Path, *, allow_remote: bool = False) -> Starlette: + """Build the ASGI app: the ``/proxy/*`` bridge + the static SPA. + + ``allow_remote`` drops the loopback guard on the bridge — only for + deliberately-exposed deployments behind their own auth. + """ + root = console_dir.resolve() + index = root / "index.html" + + async def _proxy(request: Request) -> Response: + client_host = request.client.host if request.client else None + if not allow_remote and client_host not in _LOOPBACK: + return _err(403, "forbidden", "proxy is only available to loopback clients") + + target_raw = request.headers.get("x-vouch-target") + if not target_raw: + return _err(400, "bad_target", "missing X-Vouch-Target header") + parsed = urllib.parse.urlparse(target_raw) + if parsed.scheme not in ("http", "https") or not parsed.netloc: + return _err(400, "bad_target", f"not a valid http(s) target: {target_raw}") + + # The path after /proxy is appended to the target's host:port; any path + # on the target itself is dropped, matching vouch-proxy.ts exactly. + sub = request.url.path[len("/proxy") :] or "/" + fwd_url = f"{parsed.scheme}://{parsed.netloc}{sub}" + if request.url.query: + fwd_url += f"?{request.url.query}" + + fwd_headers: dict[str, str] = {} + if request.headers.get("content-type"): + fwd_headers["content-type"] = request.headers["content-type"] + if request.headers.get("authorization"): + fwd_headers["authorization"] = request.headers["authorization"] + body = await request.body() + method = request.method + + def _do() -> tuple[int, str, bytes]: + req = urllib.request.Request( + fwd_url, data=body or None, method=method, headers=fwd_headers + ) + try: + with urllib.request.urlopen(req, timeout=_PROXY_TIMEOUT) as resp: + ctype = resp.headers.get("content-type", "application/json") + return resp.status, ctype, resp.read() + except urllib.error.HTTPError as exc: + # A backend 4xx/5xx is a real answer — pass it through unchanged + # rather than masking it as a proxy 502. + ctype = exc.headers.get("content-type", "application/json") if exc.headers else ( + "application/json" + ) + return exc.code, ctype, exc.read() + + try: + status, ctype, payload = await run_in_threadpool(_do) + except urllib.error.URLError as exc: + return _err(502, "proxy_error", str(exc.reason)) + return Response(content=payload, status_code=status, media_type=ctype) + + async def _spa(request: Request) -> Response: + """Serve a real asset, else index.html so client-side routing works.""" + rel = request.path_params.get("full_path", "") + if rel: + candidate = (root / rel).resolve() + try: + candidate.relative_to(root) # reject path-traversal escapes + except ValueError: + candidate = index + if candidate.is_file(): + return FileResponse(candidate) + return FileResponse(index) + + routes = [ + Route("/proxy", _proxy, methods=_PROXY_METHODS), + Route("/proxy/{path:path}", _proxy, methods=_PROXY_METHODS), + Route("/{full_path:path}", _spa, methods=["GET", "HEAD"]), + ] + return Starlette(routes=routes) + + +def serve_console( + *, + host: str = "127.0.0.1", + port: int = 5173, + allow_remote: bool = False, + console_dir: Path | None = None, +) -> None: + """Serve the console with uvicorn (blocks). Raises ``ConsoleError`` early + if no built SPA can be found, before uvicorn is ever started.""" + resolved = console_dir if console_dir is not None else resolve_console_dir() + if resolved is None: + raise ConsoleError( + "no built vouch console found. from a source checkout run " + "`npm run build` in webapp/; otherwise install a release wheel of " + "vouch-kb[web] (the console ships inside it)." + ) + import uvicorn + + app = build_console_app(resolved, allow_remote=allow_remote) + uvicorn.run(app, host=host, port=port, log_level="info") diff --git a/src/vouch/web/dual_solve_api.py b/src/vouch/web/dual_solve_api.py index c696cb94..e19db26b 100644 --- a/src/vouch/web/dual_solve_api.py +++ b/src/vouch/web/dual_solve_api.py @@ -71,7 +71,13 @@ def _serialize(job: DualSolveJob) -> dict[str, Any]: "log": c.log, "diff": c.diff} for c in job.candidates ], - "recommendation": ds.recommendation(job.candidates), + # No candidates yet (still running) -> no recommendation. Computing it + # over [] returns the "neither engine produced a usable diff" hint, + # which the SPA would otherwise render as a false failure for the whole + # duration of a run that is still in progress. + "recommendation": ( + ds.recommendation(job.candidates) if job.candidates else None + ), "proposed_ids": list(job.proposed_ids), "kept_branch": job.kept_branch, "changed_files": ds.changed_files(kept.diff) if kept is not None else [], diff --git a/src/vouch/web/server.py b/src/vouch/web/server.py index 597e15ba..82488545 100644 --- a/src/vouch/web/server.py +++ b/src/vouch/web/server.py @@ -39,6 +39,7 @@ import os import secrets from dataclasses import dataclass, field +from datetime import datetime from pathlib import Path from typing import Any from urllib.parse import quote @@ -585,6 +586,71 @@ async def contradict(claim_id: str, against: str = Form(...)) -> Any: await _notify("audit", action="contradict", claim_id=claim_id) return RedirectResponse(url="/audit", status_code=303) + # --- bulk-clear auto-approved claims (issue #433) --- + # + # A viewport over ``lifecycle.clear_claims``: the GET renders a dry-run + # preview of every auto-approved durable claim that matches the filter, + # and the POST archives them (never deletes) under a single + # ``claim.bulk_clear`` audit event — identical to ``vouch claims-clear``. + # Human-approved claims are never touched. ``auto_only`` is fixed on so + # the console can only ever clear the calibration cruft the feature + # targets, not reviewed knowledge. + + def _parse_before(value: str | None) -> datetime | None: + if not value: + return None + try: + return datetime.fromisoformat(value) + except ValueError as e: + raise HTTPException( + status_code=400, + detail=f"invalid date: {value} (use ISO 8601, e.g. 2026-07-01)", + ) from e + + @app.get("/clear-claims", response_class=HTMLResponse, dependencies=guarded) + def clear_claims_view( + request: Request, before: str | None = None, cleared: int | None = None, + ) -> Any: + before_err: str | None = None + candidates: list[dict[str, Any]] = [] + try: + before_dt = _parse_before(before) + preview = life.clear_claims( + store, auto_only=True, before=before_dt, + actor=reviewer(), dry_run=True, + ) + candidates = [ + { + "id": c.id, + "text": c.text[:200], + "created_at": c.created_at.isoformat(timespec="seconds"), + } + for c in preview + ] + except HTTPException as e: + before_err = str(e.detail) + return _tmpl(request, "clear.html", { + "candidates": candidates, + "count": len(candidates), + "before": before or "", + "before_err": before_err, + "cleared": cleared, + "active": "clear", + }) + + @app.post("/clear-claims", dependencies=guarded) + async def clear_claims_apply(before: str | None = Form(default=None)) -> Any: + before_dt = _parse_before(before) + cleared = await run_in_threadpool( + life.clear_claims, store, + auto_only=True, before=before_dt, actor=reviewer(), dry_run=False, + ) + await _notify("clear", action="clear_claims", count=len(cleared)) + suffix = f"&before={quote(before)}" if before else "" + return RedirectResponse( + url=f"/clear-claims?cleared={len(cleared)}{suffix}", status_code=303, + ) + # --- audit timeline --- @app.get("/audit", response_class=HTMLResponse, dependencies=guarded) diff --git a/src/vouch/web/static/app.css b/src/vouch/web/static/app.css index a69df0b3..26339ae0 100644 --- a/src/vouch/web/static/app.css +++ b/src/vouch/web/static/app.css @@ -248,3 +248,38 @@ pre { border-top: 1px solid var(--rule); margin-top: 4rem; } + +/* --- clear auto-claims (issue #433) --- */ +.clear .lede { + font-family: var(--sans); + font-size: 0.95rem; + color: var(--muted); + max-width: 60ch; + margin: 0.75rem 0 1.25rem; +} +.clear-filter { + display: flex; + align-items: center; + gap: 0.6rem; + margin: 1rem 0 1.5rem; + font-family: var(--sans); + font-size: 0.85rem; +} +.clear-filter label { color: var(--muted); } +button.filter { border-color: var(--ink); color: var(--ink); } +button.filter:hover { background: var(--ink); color: var(--paper); } +.clear-list { list-style: none; padding: 0; margin: 0; } +.clear-row { + padding: 0.75rem 0; + border-bottom: 1px solid var(--rule); +} +.clear-meta { + display: flex; + gap: 1rem; + align-items: baseline; + font-family: var(--mono); + font-size: 0.8rem; + color: var(--muted); +} +.clear-apply { margin-top: 1.5rem; } + diff --git a/src/vouch/web/static/diff_view.js b/src/vouch/web/static/diff_view.js new file mode 100644 index 00000000..c4868050 --- /dev/null +++ b/src/vouch/web/static/diff_view.js @@ -0,0 +1,90 @@ +// diff_view.js — pure helpers for the dual-solve file-changes view. +// no imports, so tests can execute the module directly under node +// (tests/test_web_diff_view.py); dual_solve.js consumes it in the browser. + +// split a unified-diff string into per-file sections with +/-/context classes. +export function parseDiff(diff) { + const files = []; + let cur = null; + for (const line of (diff || "").split("\n")) { + if (line.startsWith("diff --git")) { + const m = line.match(/ b\/(.+)$/); + cur = { path: m ? m[1] : line, lines: [] }; + files.push(cur); + } else if (!cur) { + continue; + } else if ( + // only the file-header markers, which always carry a trailing space and + // path ("+++ b/x", "--- a/x"). a content line like "++counter" or + // "---flag" must NOT be skipped. + line.startsWith("+++ ") || + line.startsWith("--- ") || + line.startsWith("index ") + ) { + continue; + } else if (line.startsWith("@@")) { + cur.lines.push({ cls: "hunk", text: line }); + } else if (line.startsWith("+")) { + cur.lines.push({ cls: "add", text: line }); + } else if (line.startsWith("-")) { + cur.lines.push({ cls: "del", text: line }); + } else { + cur.lines.push({ cls: "ctx", text: line }); + } + } + return files; +} + +function sortNodes(nodes) { + nodes.sort((a, b) => { + if (a.type !== b.type) return a.type === "tree" ? -1 : 1; + return a.name.localeCompare(b.name); + }); + for (const n of nodes) if (n.children) sortNodes(n.children); +} + +// build a nested file tree from a flat list of paths. nodes are folders-first +// then alphabetical at every level; intermediate directories are synthesized +// from path segments so a lone "a/b/c.py" still produces the a → b → c.py +// chain. +export function buildFileTree(paths) { + const roots = []; + // path -> node, so intermediate dirs are created once and reused. + const byPath = new Map(); + + for (const raw of paths) { + const full = raw.replace(/^\/+|\/+$/g, ""); + if (!full) continue; + const segs = full.split("/"); + let prefix = ""; + let siblings = roots; + + segs.forEach((seg, i) => { + prefix = prefix ? `${prefix}/${seg}` : seg; + const isLeaf = i === segs.length - 1; + let node = byPath.get(prefix); + if (!node) { + node = isLeaf + ? { name: seg, path: prefix, type: "blob" } + : { name: seg, path: prefix, type: "tree", children: [] }; + byPath.set(prefix, node); + siblings.push(node); + } + // descend; only directory nodes carry children. + if (node.children) siblings = node.children; + }); + } + + sortNodes(roots); + return roots; +} + +// flatten a tree into render-order rows for the rail: directories become +// non-interactive labels, files clickable rows; depth drives indentation. +export function flattenTree(nodes, depth = 0, out = []) { + for (const n of nodes) { + out.push({ name: n.name, path: n.path, type: n.type, depth }); + if (n.children) flattenTree(n.children, depth + 1, out); + } + return out; +} diff --git a/src/vouch/web/static/dual_solve.css b/src/vouch/web/static/dual_solve.css index e77e4bf8..c4399d69 100644 --- a/src/vouch/web/static/dual_solve.css +++ b/src/vouch/web/static/dual_solve.css @@ -16,3 +16,14 @@ .ln-hunk { color:#06c; display:block; } .ln-ctx { display:block; } .ds-choose, .ds-result { margin:1rem 0; display:flex; gap:.5rem; align-items:center; flex-wrap:wrap; } + +/* file-changes view (candidate pane): tree rail + single-file diff pane */ +.fc { display:flex; gap:8px; align-items:stretch; max-height:460px; } +.fc-rail { flex:0 0 150px; max-width:150px; overflow:auto; font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace; font-size:12px; padding-right:4px; border-right:1px solid #ddd; } +.fc-pane { flex:1 1 auto; min-width:0; overflow:auto; } +.fc-dir { color:#777; padding:1px 2px; white-space:nowrap; overflow:hidden; text-overflow:ellipsis; } +.fc-file { display:block; width:100%; text-align:left; background:none; border:0; font:inherit; color:inherit; padding:1px 4px; white-space:nowrap; overflow:hidden; text-overflow:ellipsis; cursor:pointer; border-radius:3px; } +.fc-file:hover { background:#f2f2f2; } +.fc-file:focus-visible { outline:none; box-shadow: inset 0 0 0 1px #06c; } +.fc-file.sel { background:#e6f0fa; } +.fc-empty { color:#777; font-size:12.5px; } diff --git a/src/vouch/web/static/dual_solve.js b/src/vouch/web/static/dual_solve.js index a7772013..4ce26363 100644 --- a/src/vouch/web/static/dual_solve.js +++ b/src/vouch/web/static/dual_solve.js @@ -1,31 +1,5 @@ import { ref, reactive, onMounted } from "/static/vendor/vue.esm-browser.prod.js"; - -// Minimal unified-diff parser: groups lines per file with +/-/context classes. -function parseDiff(diff) { - const files = []; - let cur = null; - for (const line of (diff || "").split("\n")) { - if (line.startsWith("diff --git")) { - const m = line.match(/ b\/(.+)$/); - cur = { path: m ? m[1] : line, lines: [] }; - files.push(cur); - } else if (!cur) { - continue; - } else if (line.startsWith("+++") || line.startsWith("---") || - line.startsWith("index ")) { - continue; - } else if (line.startsWith("@@")) { - cur.lines.push({ cls: "hunk", text: line }); - } else if (line.startsWith("+")) { - cur.lines.push({ cls: "add", text: line }); - } else if (line.startsWith("-")) { - cur.lines.push({ cls: "del", text: line }); - } else { - cur.lines.push({ cls: "ctx", text: line }); - } - } - return files; -} +import { parseDiff, buildFileTree, flattenTree } from "/static/diff_view.js"; export default { setup() { @@ -39,12 +13,33 @@ export default { changed_files: [], recommendation: null, }); + // engine -> selected file path in that candidate's file-changes view. + // selection is per-candidate by design: picking a file in the claude pane + // never moves the codex pane. lives outside applyState so polling doesn't + // reset it. + const selected = reactive({}); + function applyState(s) { Object.assign(job, s, { - candidates: (s.candidates || []).map( - (c) => ({ ...c, files: parseDiff(c.diff) })), + candidates: (s.candidates || []).map((c) => { + const files = parseDiff(c.diff); + return { + ...c, + files, + rows: flattenTree(buildFileTree(files.map((f) => f.path))), + }; + }), }); } + // selected file object for a candidate; falls back to the first changed + // file when nothing is selected yet or the diff no longer has the path. + function activeFile(c) { + const want = selected[c.engine]; + return c.files.find((f) => f.path === want) || c.files[0] || null; + } + function selectFile(engine, path) { + selected[engine] = path; + } async function refresh() { if (!job.id) return; const r = await fetch(`/dual-solve/job/${job.id}`); @@ -90,7 +85,10 @@ export default { } onMounted(connectWs); - return { issueUrl, claudeEffort, codexEffort, reason, job, run, choose }; + return { + issueUrl, claudeEffort, codexEffort, reason, job, run, choose, + activeFile, selectFile, + }; }, template: `
@@ -117,17 +115,29 @@ export default {

{{c.engine}} {{c.branch}}

failed: {{c.error}}

-
    -
  • {{f}}
  • -
{{c.engine}} log
{{c.log}}
-
-
{{f.path}}
-
{{l.text}}\\n
+
+
+ +
+
+
+
{{activeFile(c).path}}
+
{{l.text + '\\n'}}
+
+
+

(no file changes)

diff --git a/src/vouch/web/templates/base.html b/src/vouch/web/templates/base.html index eafa36a1..e3f92ff0 100644 --- a/src/vouch/web/templates/base.html +++ b/src/vouch/web/templates/base.html @@ -16,6 +16,7 @@
+ + + + + + + diff --git a/webapp/tsconfig.json b/webapp/tsconfig.json new file mode 100644 index 00000000..12058f97 --- /dev/null +++ b/webapp/tsconfig.json @@ -0,0 +1,18 @@ +{ + "compilerOptions": { + "target": "ES2022", + "lib": ["ES2022", "DOM", "DOM.Iterable"], + "module": "ESNext", + "moduleResolution": "bundler", + "jsx": "react-jsx", + "strict": true, + "noUnusedLocals": true, + "noUnusedParameters": true, + "noFallthroughCasesInSwitch": true, + "skipLibCheck": true, + "isolatedModules": true, + "noEmit": true, + "types": ["vite/client", "node"] + }, + "include": ["src", "plugins", "e2e", "vite.config.ts", "vitest.config.ts", "playwright.config.ts"] +} diff --git a/webapp/vite.config.ts b/webapp/vite.config.ts new file mode 100644 index 00000000..de6acf2b --- /dev/null +++ b/webapp/vite.config.ts @@ -0,0 +1,9 @@ +import { defineConfig } from 'vite' +import react from '@vitejs/plugin-react' +import tailwindcss from '@tailwindcss/vite' +import { vouchProxy } from './plugins/vouch-proxy' +import { claudeBridge } from './plugins/claude-bridge' + +export default defineConfig({ + plugins: [react(), tailwindcss(), vouchProxy(), claudeBridge()], +}) diff --git a/webapp/vitest.config.ts b/webapp/vitest.config.ts new file mode 100644 index 00000000..75196c01 --- /dev/null +++ b/webapp/vitest.config.ts @@ -0,0 +1,11 @@ +import { defineConfig } from 'vitest/config' +import react from '@vitejs/plugin-react' + +export default defineConfig({ + plugins: [react()], + test: { + environment: 'jsdom', + setupFiles: ['./src/test/setup.ts'], + include: ['src/**/*.test.{ts,tsx}', 'plugins/**/*.test.ts'], + }, +})