From 556f8470642218ad2235d73efe0511f1e1587289 Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 01:44:20 +0800 Subject: [PATCH] feat(acp-snapshot): extract the ACP snapshot suite into a support package MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The snapshot tier's machinery leaves examples/acp-agent/tests for packages/support/acp-snapshot (@deepseek-ai/dsh-acp-snapshot), where the coverage gate measures it and a second example can consume it instead of forking it: harness.ts (runScenario, parameterized by an AgentUnderTest {binScript, configPath, tsconfigPath} instead of module constants), normalize.ts (moved verbatim), and suite.ts (defineAcpSnapshotSuite — the per-scenario golden/log compares, record write-back, per-suite header pin with its uniformity guard, and the fixture guard block, lifted from acp.snapshot.ts). The example file collapses to its scenario table plus one factory call; env reading (DSH_SNAPSHOT) stays at that edge. The exactly-one-pin meta-test generalizes from the hardcoded text-turn name to "exactly one per suite" — which scenario pins is the scenario table's reviewable choice (per-suite pinning per the proposal RFC). Extraction parity: pnpm run test:snapshot is 36 passed + fs-policy-reject failing BEFORE AND AFTER (BSD-sed environment failure, reproduced at the base commit in a clean worktree — the recorded golden's sed -i syntax is GNU-only), with zero byte changes under examples/acp-agent/tests/snapshots/. Coverage for the new src files lands in the next commit. --- docs/config-catalog.md | 1 + docs/module-graph.md | 2 + ...0-remove-redundant-snapshot-log-goldens.md | 2 +- ...-request-header-content-in-one-scenario.md | 2 +- .../2026-07-08-shared-acp-snapshot-package.md | 2 +- docs/testing.md | 2 +- examples/acp-agent/tests/acp.e2e.ts | 4 +- examples/acp-agent/tests/acp.snapshot.ts | 331 +--------------- knip.json | 7 +- packages/support/README.md | 3 +- packages/support/acp-snapshot/README.md | 36 ++ packages/support/acp-snapshot/package.json | 35 ++ .../support/acp-snapshot/src/harness.ts | 79 ++-- packages/support/acp-snapshot/src/index.ts | 37 ++ .../support/acp-snapshot/src/normalize.ts | 22 +- packages/support/acp-snapshot/src/suite.ts | 355 ++++++++++++++++++ .../acp-snapshot/tests/normalize.spec.ts | 4 +- packages/support/acp-snapshot/tsconfig.json | 11 + pnpm-lock.yaml | 68 ++++ tsconfig.build.json | 1 + tsconfig.json | 1 + 21 files changed, 654 insertions(+), 351 deletions(-) create mode 100644 packages/support/acp-snapshot/README.md create mode 100644 packages/support/acp-snapshot/package.json rename examples/acp-agent/tests/snapshot-harness.ts => packages/support/acp-snapshot/src/harness.ts (86%) create mode 100644 packages/support/acp-snapshot/src/index.ts rename examples/acp-agent/tests/snapshot-normalize.ts => packages/support/acp-snapshot/src/normalize.ts (91%) create mode 100644 packages/support/acp-snapshot/src/suite.ts rename examples/acp-agent/tests/snapshot-normalize.spec.ts => packages/support/acp-snapshot/tests/normalize.spec.ts (98%) create mode 100644 packages/support/acp-snapshot/tsconfig.json diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 2c101e318e..e268f63da6 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -810,6 +810,7 @@ Abstract service classes — a deployment loads a concrete implementation packag Imported as libraries by other packages; a `cordis.yml` cannot load them. +- `@deepseek-ai/dsh-acp-snapshot` ([`packages/support/acp-snapshot/src/index.ts`](../packages/support/acp-snapshot/src/index.ts)) - `@deepseek-ai/dsh-app-boot` ([`packages/ui/app-boot/src/index.ts`](../packages/ui/app-boot/src/index.ts)) - `@deepseek-ai/dsh-brand` ([`packages/util/brand/src/index.ts`](../packages/util/brand/src/index.ts)) - `@deepseek-ai/dsh-hook-protocol` ([`packages/hooks/hook-protocol/src/index.ts`](../packages/hooks/hook-protocol/src/index.ts)) diff --git a/docs/module-graph.md b/docs/module-graph.md index 8283a8b756..c08b1fd171 100644 --- a/docs/module-graph.md +++ b/docs/module-graph.md @@ -68,6 +68,7 @@ flowchart TD pkg_session_persistence_sqlite["session-persistence-sqlite"] end subgraph group_support["packages/support"] + pkg_acp_snapshot["acp-snapshot"] pkg_invariants["invariants"] pkg_llm_replay["llm-replay"] pkg_subagent_mock["subagent-mock"] @@ -207,6 +208,7 @@ flowchart TD | Package | Group | Depends on | | --- | --- | --- | | [`brand`](../packages/util/brand) | `util` | — | +| [`acp-snapshot`](../packages/support/acp-snapshot) | `support` | — | | [`app-boot`](../packages/ui/app-boot) | `ui` | — | | [`llm`](../packages/llm/llm) | `llm` | [`brand`](../packages/util/brand) | | [`bash`](../packages/bash/bash) | `bash` | [`brand`](../packages/util/brand) | diff --git a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md index 07a1cfd731..2a6f8b7ae3 100644 --- a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md +++ b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md @@ -32,4 +32,4 @@ Reviewers lose one artifact name that made the expected persisted log visually s ## Implementation note -The comparison normalizes BOTH sides, but each against its OWN volatile values, not a shared context. A raw harvested `session.jsonl` bakes in the recording run's session id, cwd, and timestamps; the replay run produces fresh ones. `normalizeSessionLog` scrubs cwd by exact string match, so normalizing the fixture against the *replay* run's cwd would leave the recorded cwd in the header unscrubbed and the compare would fail. The harness therefore derives the fixture's normalize context from its OWN header line (`{ type:'session', id, cwd }`) — `fixtureContext()` in `acp.snapshot.ts` — so both sides scrub to the same `{{sessionId}}`/`{{cwd}}` tokens. An authored fixture copied from the old golden already carries the normalized header (`id:'{{sessionId}}'`, `cwd:'{{cwd}}'`), which yields those tokens as the volatile values and scrubs idempotently. The session-log side uses a plain normalized-string `toEqual`, NOT `toMatchFileSnapshot`, so a run never overwrites the fixture. +The comparison normalizes BOTH sides, but each against its OWN volatile values, not a shared context. A raw harvested `session.jsonl` bakes in the recording run's session id, cwd, and timestamps; the replay run produces fresh ones. `normalizeSessionLog` scrubs cwd by exact string match, so normalizing the fixture against the *replay* run's cwd would leave the recorded cwd in the header unscrubbed and the compare would fail. The harness therefore derives the fixture's normalize context from its OWN header line (`{ type:'session', id, cwd }`) — `fixtureContext()` in `dsh-acp-snapshot`'s suite module — so both sides scrub to the same `{{sessionId}}`/`{{cwd}}` tokens. An authored fixture copied from the old golden already carries the normalized header (`id:'{{sessionId}}'`, `cwd:'{{cwd}}'`), which yields those tokens as the volatile values and scrubs idempotently. The session-log side uses a plain normalized-string `toEqual`, NOT `toMatchFileSnapshot`, so a run never overwrites the fixture. diff --git a/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md b/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md index 4e4b6a85d7..862dfd42fa 100644 --- a/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md +++ b/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md @@ -8,7 +8,7 @@ Every model-driving ACP snapshot fixture (`session.jsonl`) embedded the full com ## Decision -Exactly one scenario — `text-turn`, flagged `pinsHeader` in `acp.snapshot.ts` — commits and compares the full request-header content. Every other fixture stores and compares that content as stable tokens via the pure normalizer `scrubRequestHeaders` in `snapshot-normalize.ts`: a `request/header` event's `header.system` becomes `"{{system}}"` and `header.tools` becomes `"{{tools}}"`; a `request/header-delta` keeps its structural facts — the system delta's `keepStart`/`keepEnd` line positions with one `{{system}}` token per inserted line, the tools delta's added/removed/changed tool names — and tokenizes only the bulk (prompt text, schema bodies), so two different deltas still compare different. The scrub is composed in front of `normalizeSessionLog` on BOTH sides of a non-pinning scenario's log compare and applied to the harvested logs record mode writes, so a re-record cannot smuggle the content back. Absent fields stay absent — WHETHER a header carried a prompt or tools is behavior and stays visible — and `config`/`reason` stay verbatim: a model swap churns every fixture by design (it invalidates the recorded responses), while a prompt or schema edit churns none of them (replay derives model behavior exclusively from `assistant/chunk` events and never reads header content — see `dsh-llm-replay`). +Exactly one scenario — `text-turn`, flagged `pinsHeader` in the `acp.snapshot.ts` scenario table — commits and compares the full request-header content; the pin mechanics live in [`dsh-acp-snapshot`](../../../../packages/support/acp-snapshot/README.md), whose suite factory enforces one pin per consuming suite. Every other fixture stores and compares that content as stable tokens via the pure normalizer `scrubRequestHeaders` in that package's `normalize.ts`: a `request/header` event's `header.system` becomes `"{{system}}"` and `header.tools` becomes `"{{tools}}"`; a `request/header-delta` keeps its structural facts — the system delta's `keepStart`/`keepEnd` line positions with one `{{system}}` token per inserted line, the tools delta's added/removed/changed tool names — and tokenizes only the bulk (prompt text, schema bodies), so two different deltas still compare different. The scrub is composed in front of `normalizeSessionLog` on BOTH sides of a non-pinning scenario's log compare and applied to the harvested logs record mode writes, so a re-record cannot smuggle the content back. Absent fields stay absent — WHETHER a header carried a prompt or tools is behavior and stays visible — and `config`/`reason` stay verbatim: a model swap churns every fixture by design (it invalidates the recorded responses), while a prompt or schema edit churns none of them (replay derives model behavior exclusively from `assistant/chunk` events and never reads header content — see `dsh-llm-replay`). A system-prompt or tool-schema change therefore lands as exactly one committed-fixture diff — the pinned `text-turn` header line — updated by hand or by re-recording that one scenario (`pnpm run test:snapshot:record` with `-t text-turn`). diff --git a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md index 2bd061d422..839e1d4542 100644 --- a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md +++ b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md @@ -4,7 +4,7 @@ Status: proposed ## Problem -The ACP snapshot tier ([snapshot RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md)) is built from three modules that live inside one example's test directory: [snapshot-harness.ts](../../../../examples/acp-agent/tests/snapshot-harness.ts) (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), [snapshot-normalize.ts](../../../../examples/acp-agent/tests/snapshot-normalize.ts) (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in [acp.snapshot.ts](../../../../examples/acp-agent/tests/acp.snapshot.ts) (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). +The ACP snapshot tier ([snapshot RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md)) is built from three modules that live inside one example's test directory: `snapshot-harness.ts` (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), `snapshot-normalize.ts` (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in [acp.snapshot.ts](../../../../examples/acp-agent/tests/acp.snapshot.ts) (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). A second ACP example that wants snapshot coverage — the sandbox/approval composition is the immediate consumer — can only copy those modules, forking exactly the logic that must not drift: record write-back, header scrubbing, child-session harvest ordering. The spawn/client glue is already triplicated across [acp.e2e.ts](../../../../examples/acp-agent/tests/acp.e2e.ts), [hooks.e2e.ts](../../../../examples/acp-agent/tests/hooks.e2e.ts), and the harness, marked by `TODO(acp-test-harness)`. diff --git a/docs/testing.md b/docs/testing.md index d7ed14ecb9..a8cf977e85 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -30,4 +30,4 @@ An e2e assertion re-runs the command or re-reads the file externally; a keyword ## When a snapshot test is required -Any change affecting the editor-facing transcript or end-to-end agent UX — the ACP bridge, the loop's observable output, tool presentation — adds or updates a scenario under `examples/acp-agent/tests/snapshots/` (or states in the PR why none applies). New capability seams, lifecycle shapes, or transcript surfaces name their coverage at every tier at plan time and verify the harness can express it — a harness gap is scheduled work, not a mid-build surprise. +Any change affecting the editor-facing transcript or end-to-end agent UX — the ACP bridge, the loop's observable output, tool presentation — adds or updates a scenario in the owning example's snapshot suite (`examples//tests/snapshots/`, a scenario table over the [`dsh-acp-snapshot`](../packages/support/acp-snapshot/README.md) suite factory; `examples/acp-agent` is the primary suite), or states in the PR why none applies. New capability seams, lifecycle shapes, or transcript surfaces name their coverage at every tier at plan time and verify the harness can express it — a harness gap is scheduled work, not a mid-build surprise. diff --git a/examples/acp-agent/tests/acp.e2e.ts b/examples/acp-agent/tests/acp.e2e.ts index 3245fe1750..714791fa3e 100644 --- a/examples/acp-agent/tests/acp.e2e.ts +++ b/examples/acp-agent/tests/acp.e2e.ts @@ -57,8 +57,8 @@ interface Spawned { } // TODO(acp-test-harness): this subprocess/client boot glue is duplicated with -// hooks.e2e.ts and partly with snapshot-harness.ts. Extract one shared ACP test -// launcher before the TSX/env/permission-stub details drift again. +// hooks.e2e.ts and partly with dsh-acp-snapshot's harness. Migrate both e2e +// files onto that launcher before the TSX/env/permission-stub details drift. function spawnAcpAgent(cwd: string, env: NodeJS.ProcessEnv = process.env): Spawned { const child = spawn( process.execPath, diff --git a/examples/acp-agent/tests/acp.snapshot.ts b/examples/acp-agent/tests/acp.snapshot.ts index bde4f4dbb9..b864189a61 100644 --- a/examples/acp-agent/tests/acp.snapshot.ts +++ b/examples/acp-agent/tests/acp.snapshot.ts @@ -1,86 +1,24 @@ -import { readFile, readdir, writeFile } from 'node:fs/promises' -import { existsSync } from 'node:fs' import { fileURLToPath } from 'node:url' import { dirname, join } from 'node:path' -import { describe, expect, it } from 'vitest' -import { type HarvestedLog, type InputScript, runScenario } from './snapshot-harness.ts' -import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from './snapshot-normalize.ts' +import { defineAcpSnapshotSuite, type Scenario } from '@deepseek-ai/dsh-acp-snapshot' /** - * ACP snapshot tests (REPLAY by default, keyless). Each scenario under - * `snapshots//` ships an `input.json` (the client stdin script) and a - * `session.jsonl` fixture; replay boots the real acp-agent subprocess, drives - * it, and diffs the normalized stdout transcript against the committed - * `stdout.golden.jsonl`. For model scenarios it ALSO checks the re-persisted - * session log — against the `session.jsonl` fixture itself, not a separate - * golden: the fixture doubles as the replay source (recorded scenarios) and the - * expected produced log (both sides normalized before comparing). - * - * Request-header content (the composed system prompt + tool schemas riding on - * `request/header` events) is pinned by exactly ONE scenario — the one with - * `pinsHeader` — and scrubbed to `{{system}}`/`{{tools}}` tokens in every - * other fixture and compare, so a prompt or tool-schema edit churns one - * committed line instead of every fixture. A per-run uniformity guard keeps - * the single pin sound: every live header must equal the pinned one, and no - * header-delta may appear outside the pinning scenario (see the - * pinned-header RFC, - * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). - * - * `pnpm run test:snapshot:record` (DSH_SNAPSHOT=record + -u) re-records the - * `session.jsonl` fixtures against the real API and refreshes the stdout golden - * in one pass. + * The acp-agent example's snapshot suite: the scenario table for + * `dsh-acp-snapshot`'s suite factory, which owns every compare/guard mechanic + * (golden + re-persisted-log diffs, record write-back, the pinned-header + * uniformity guard, the fixture guards). Fixtures live under `snapshots//`; + * `pnpm run test:snapshot:record` re-records the `recorded` scenarios against + * the real API. See the package README (packages/support/acp-snapshot) and the + * snapshot RFC, docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. */ -const SNAPSHOTS_DIR = join(dirname(fileURLToPath(import.meta.url)), 'snapshots') -const RECORDING = process.env.DSH_SNAPSHOT === 'record' - -/** A snapshot scenario and how its fixtures are produced. */ -interface Scenario { - name: string - /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ - hasModelTurn: boolean - /** - * Whether the run persists a comparable session log to diff against the - * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn - * always produces a log worth comparing). Set it independently for a scenario - * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked - * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` - * events but never calls the model. - */ - comparesLog?: boolean - /** - * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` - * from the LIVE API. `recorded` scenarios are model-driven and reproducible; - * `authored` scenarios (a hand-written `replay.override.json` sidecar drives - * replay — e.g. a provider error or a cancel, which the live API can't be - * coaxed into deterministically — or a deterministic hook scenario whose - * derived empty script needs no sidecar) are NEVER re-recorded. - */ - recorded: boolean - /** - * How many SUBAGENT child sessions this scenario records beyond the top-level - * one (0 for a single-session scenario). Each child rides in a sibling fixture - * `session..jsonl` (1-based); replay forwards them to `dsh-llm-replay` so - * each child session replays from its own script, and record mode writes the - * harvested child logs back to those files. Defaults to 0. - */ - childSessions?: number - /** - * Whether THIS scenario's fixtures keep the full request-header content (the - * composed system prompt and tool schema list on `request/header` / - * `request/header-delta` events) and compare it verbatim. Exactly one - * scenario pins it; every other scenario stores and compares that content as - * `{{system}}`/`{{tools}}` tokens ({@link scrubRequestHeaders}), so a system - * prompt or tool-schema change shows up as ONE committed-fixture diff, not - * one per scenario. One pin suffices because header composition is - * suite-uniform (parent, spawn child, and fork child all compose the same - * prompt-modulo-cwd and the same tools) — and that premise is ASSERTED, not - * assumed: every non-pinning run's live headers must equal the pinned - * fixture's (normalized), so a session-dependent header (say, a restricted - * subagent toolset) fails loud until it gets its own pinning scenario. - * Defaults to false. - */ - pinsHeader?: boolean +// The dsh-acp-agent bin (the demo:acp entry), this example's cordis.yml, and +// the repo-root tsconfig (four levels up from examples/acp-agent/tests) — all +// ABSOLUTE: the subprocess cwd is a temp dir outside the repo. +const AGENT = { + binScript: fileURLToPath(new URL('../../../packages/ui/acp-agent/src/bin.ts', import.meta.url)), + configPath: fileURLToPath(new URL('../cordis.yml', import.meta.url)), + tsconfigPath: fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)), } const SCENARIOS: Scenario[] = [ @@ -148,238 +86,9 @@ const SCENARIOS: Scenario[] = [ { name: 'hook-codex-stop-continue', hasModelTurn: true, recorded: true }, ] -/** The single header-pinning scenario. Guarded here (and by a meta-test) so the pin cannot silently vanish. */ -const pinningScenario = SCENARIOS.find(s => s.pinsHeader === true) -if (pinningScenario === undefined) throw new Error('acp.snapshot: no scenario pins the request-header content') - -/** The sibling child-fixture paths for a scenario (`session.1.jsonl` …). */ -function childFixturePaths(dir: string, childSessions: number): string[] { - return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) -} - -/** - * Derive the {@link NormalizeContext} for a `session.jsonl` fixture from its own - * header line (`{ type: 'session', id, cwd }`). A committed fixture carries the - * session id and cwd of the run that harvested it — different from the live - * replay run — so normalizing it against the live run's ctx would leave those - * recorded values unscrubbed. Reading them from the header scrubs the fixture's - * own id/cwd to the same `{{sessionId}}`/`{{cwd}}` tokens the replay output gets. - * An authored fixture whose header is already normalized (`id:'{{sessionId}}'`, - * `cwd:'{{cwd}}'`) yields those tokens as the volatile values, so scrubbing them - * is an idempotent no-op. A header with no `cwd` falls back to a sentinel that - * cannot occur in a log (NOT `''`, which `String.split` would match on every - * character boundary and corrupt the output). - */ -function fixtureContext(fixture: string): NormalizeContext { - const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' - const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } - return { - sessionIds: typeof header.id === 'string' ? [header.id] : [], - cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', - } -} - -/** - * The `data.header` payload of every `request/header` event in a session - * JSONL, in log order, with the log's volatile values scrubbed first - * ({@link normalizeSessionLog}) so headers harvested from different runs — - * each embedding its own temp cwd in the composed prompt — compare on equal - * footing. - */ -function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { - return normalizeSessionLog(rawLog, ctx) - .split('\n') - .filter(line => line.trim().length > 0) - .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) - .filter(record => record.type === 'request/header') - .map(record => record.data?.header) -} - -/** Count the `request/header-delta` events in a session JSONL. */ -function headerDeltaCount(rawLog: string): number { - return rawLog.split('\n') - .filter(line => line.trim().length > 0) - .filter(line => (JSON.parse(line) as { type?: unknown }).type === 'request/header-delta') - .length -} - -for (const scenario of SCENARIOS) { - describe(`snapshot: ${scenario.name}`, () => { - // In RECORD mode, only re-run the `recorded` (live-API) scenarios; the - // `authored` ones (sidecar-driven errors/cancel) are never re-recorded. - it.skipIf(RECORDING && !scenario.recorded)('matches the goldens', async () => { - const dir = join(SNAPSHOTS_DIR, scenario.name) - const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript - const overrideFile = join(dir, 'replay.override.json') - const workspaceDir = join(dir, 'workspace') - const childSessions = scenario.childSessions ?? 0 - const result = await runScenario(input, { - mode: RECORDING ? 'record' : 'replay', - fixtureFile: join(dir, 'session.jsonl'), - ...existsSync(overrideFile) ? { overrideFile } : {}, - // In REPLAY, forward the recorded child fixtures so each subagent session - // replays from its own script. In RECORD they are harvested, not read. - ...!RECORDING && childSessions > 0 ? { childFiles: childFixturePaths(dir, childSessions) } : {}, - ...existsSync(workspaceDir) ? { workspaceDir } : {}, - }) - - // Scrub every volatile id the run produced: the ACP server-issued session - // id plus every harvested log's recorded id (a subagent child id never - // surfaces over ACP, but it appears in the child's own log header). The - // normalizer's UUID catch-all covers any we don't enumerate. - const ctx: NormalizeContext = { - sessionIds: [ - ...result.sessionId !== undefined ? [result.sessionId] : [], - ...result.sessionLogs.map(l => l.id), - ], - cwd: result.cwd, - } - - // RECORD mode (recorded model scenarios only): persist the freshly-harvested - // logs back to their fixtures — the primary to session.jsonl, each child to - // session..jsonl in harvest order. `--update` refreshes the Vitest - // goldens but NOT these fixtures, so write them here. A non-pinning - // scenario's fixtures are written header-scrubbed, so a re-record can - // never smuggle the full prompt/schema content back into every fixture. - const scrub = scenario.pinsHeader === true - ? (log: string): string => log - : scrubRequestHeaders - if (RECORDING && scenario.recorded && scenario.hasModelTurn) { - expect(result.sessionLogs.length, 'record produced no session log to harvest').toBeGreaterThan(0) - expect(result.sessionLogs.length, `expected ${childSessions + 1} session logs (parent + children)`) - .toBe(childSessions + 1) - await writeFile(join(dir, 'session.jsonl'), scrub((result.sessionLogs[0] as HarvestedLog).content)) - for (let i = 1; i < result.sessionLogs.length; i++) { - await writeFile(join(dir, `session.${i}.jsonl`), scrub((result.sessionLogs[i] as HarvestedLog).content)) - } - } - - await expect(normalizeStdout(result.rawStdout, ctx)) - .toMatchFileSnapshot(join(dir, 'stdout.golden.jsonl')) - - // A model turn always produces a log worth comparing; a hook scenario can - // produce one without a model turn (a `rejected` turn carrying `hook/*`). - const comparesLog = scenario.comparesLog ?? scenario.hasModelTurn - if (comparesLog) { - // The harvested logs (primary-first) must match their committed fixtures - // 1:1. Each side passes through normalizeSessionLog, scrubbed against ITS - // OWN volatile values — the live run's via `ctx`, the committed fixture's - // via its own header (a committed file cannot share the live run's ids). - // Unless this scenario pins the header, both sides ALSO pass through - // scrubRequestHeaders: the live log carries the real prompt/schemas, the - // fixture carries the `{{system}}`/`{{tools}}` tokens, and the scrub is - // idempotent — so the compare checks the header's presence, position, - // reason, and config, but not its bulk content (pinned once, in the - // `pinsHeader` scenario). - expect(result.sessionLogs.length, 'this scenario must persist a session log').toBe(childSessions + 1) - const fixtureFiles = ['session.jsonl', ...Array.from({ length: childSessions }, (_, i) => `session.${i + 1}.jsonl`)] - for (let i = 0; i < fixtureFiles.length; i++) { - const harvested = scrub((result.sessionLogs[i] as HarvestedLog).content) - const fixture = scrub(await readFile(join(dir, fixtureFiles[i] as string), 'utf8')) - expect(normalizeSessionLog(harvested, ctx), `${fixtureFiles[i]} mismatch`) - .toEqual(normalizeSessionLog(fixture, fixtureContext(fixture))) - } - } - - // Header-uniformity guard: the single pin is sound only while every - // session in the suite composes the SAME header and keeps it for the - // whole run. Assert both halves live. (1) Every request/header the run - // produced (parent, spawn child, fork child, initial or resume) must - // equal the pinned fixture's header after each side is normalized - // against its own volatile values. (2) No request/header-delta may - // appear at all — a mid-run header change diverges from the pin by - // construction, and its content would be invisible under the scrub. If - // either fails, either the header changed (update the pin: re-record or - // hand-edit the pinning scenario's fixture) or composition became - // session-dependent by design (give the divergent shape its own - // pinning scenario). - if (scenario.pinsHeader !== true) { - const pinnedFixture = await readFile(join(SNAPSHOTS_DIR, pinningScenario.name, 'session.jsonl'), 'utf8') - const pinned = normalizedHeaders(pinnedFixture, fixtureContext(pinnedFixture)) - expect(pinned.length, `the pinning fixture (${pinningScenario.name}) must carry exactly one request/header`) - .toBe(1) - for (const log of result.sessionLogs) { - expect(headerDeltaCount(log.content), `session ${log.id}: a request/header-delta in a non-pinning scenario`) - .toBe(0) - const headers = normalizedHeaders(log.content, ctx) - for (const [k, header] of headers.entries()) { - expect(header, `session ${log.id}: request/header #${k + 1} diverged from the pinned (${pinningScenario.name}) header`) - .toEqual(pinned[0]) - } - } - } - }) - }) -} - -describe('snapshot fixtures', () => { - it('every scenario directory is registered (no orphans)', async () => { - // toMatchFileSnapshot does not prune orphaned golden/fixture files, so a - // renamed/removed scenario could leave a stale dir that nothing exercises. - // Fail loud on any snapshots/ not present in SCENARIOS. - const entries = await readdir(SNAPSHOTS_DIR, { withFileTypes: true }) - const onDisk = entries.filter(e => e.isDirectory()).map(e => e.name).sort() - const registered = SCENARIOS.map(s => s.name).sort() - expect(onDisk).toEqual(registered) - }) - - it('every registered scenario has its required fixture files', async () => { - // Every scenario has an input script and an stdout golden. EVERY scenario - // also needs `session.jsonl`: the harness boots `llm-replay` with that path - // as the replay source for ALL scenarios (acp.snapshot.ts passes - // `fixtureFile: /session.jsonl` unconditionally), and `loadReplayScript` - // throws "fixture not found" when it is absent and no override replaces it. - // A no-model scenario ships a header-only `session.jsonl` (it derives to an - // empty script — no model call is made); a model scenario's fixture also - // doubles as the expected-log artifact the run is diffed against. An authored - // (non-`recorded`) model scenario additionally ships a `replay.override.json` - // sidecar for the throw/hang cases a derived script cannot express. - for (const { name, hasModelTurn, recorded, childSessions } of SCENARIOS) { - const dir = join(SNAPSHOTS_DIR, name) - expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true) - expect(existsSync(join(dir, 'stdout.golden.jsonl')), `${name}/stdout.golden.jsonl`).toBe(true) - expect(existsSync(join(dir, 'session.jsonl')), `${name}/session.jsonl`).toBe(true) - if (hasModelTurn && !recorded) { - expect(existsSync(join(dir, 'replay.override.json')), `${name}/replay.override.json`).toBe(true) - } - // A nested-agent scenario ships one child fixture per recorded subagent - // session (`session.1.jsonl` …), the replay source for that child session. - for (const childFixture of childFixturePaths(dir, childSessions ?? 0)) { - expect(existsSync(childFixture), childFixture).toBe(true) - } - } - }) - - it('exactly one scenario pins the request-header content', () => { - // Zero pins would drop the prompt/schema surface from the suite entirely; - // two would split it. The single pin is the design (pinned-header RFC). - expect(SCENARIOS.filter(s => s.pinsHeader === true).map(s => s.name)).toEqual(['text-turn']) - }) - - it('committed fixtures carry request-header content ONLY in the pinning scenario', async () => { - // The whole point of the pin: a system-prompt or tool-schema change must - // churn exactly one committed line. A non-pinning fixture that carries the - // full header (a hand-recorded file, or a header line hand-edited out of - // its canonical JSON form) silently reopens the suite-wide churn, so fail - // loud here: every non-pinning session*.jsonl must be a fixed point of - // scrubRequestHeaders (apply the scrub to fix a violation), and the - // pinning scenario's fixtures must NOT be (their content IS the pin). - for (const scenario of SCENARIOS) { - const dir = join(SNAPSHOTS_DIR, scenario.name) - const files = [ - 'session.jsonl', - ...Array.from({ length: scenario.childSessions ?? 0 }, (_, i) => `session.${i + 1}.jsonl`), - ] - for (const file of files) { - const fixture = await readFile(join(dir, file), 'utf8') - if (scenario.pinsHeader === true) { - expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} must PIN the full header content`) - .not.toEqual(fixture) - } else { - expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} carries unscrubbed header content`) - .toEqual(fixture) - } - } - } - }) +defineAcpSnapshotSuite({ + agent: AGENT, + snapshotsDir: join(dirname(fileURLToPath(import.meta.url)), 'snapshots'), + scenarios: SCENARIOS, + mode: process.env.DSH_SNAPSHOT === 'record' ? 'record' : 'replay', }) diff --git a/knip.json b/knip.json index 89cd2fffe7..b5ff3c2e5d 100644 --- a/knip.json +++ b/knip.json @@ -9,7 +9,7 @@ "examples/echo-agent/tests/**/*.e2e.ts", "examples/coding-agent/tests/**/*.e2e.ts", "examples/acp-agent/tests/**/*.e2e.ts", - "examples/acp-agent/tests/**/*.snapshot.ts" + "examples/*/tests/**/*.snapshot.ts" ], "project": ["scripts/**/*.ts", "examples/**/*.ts"] }, @@ -21,6 +21,11 @@ "project": ["src/**/*.ts"], "ignoreDependencies": ["cordis"] }, + "packages/support/acp-snapshot": { + "entry": ["tests/**/*.spec.ts"], + "project": ["src/**/*.ts", "tests/**/*.ts"], + "ignoreDependencies": ["cordis"] + }, "packages/core/agent-loop": { "entry": ["tests/**/*.spec.ts", "tests/**/*.e2e.ts"], "project": ["src/**/*.ts", "tests/**/*.ts"] diff --git a/packages/support/README.md b/packages/support/README.md index 233a32d77f..2a08063bad 100644 --- a/packages/support/README.md +++ b/packages/support/README.md @@ -4,8 +4,9 @@ Packages that exist to serve development, testing, and the examples rather than | Package | Role | ctx key | |---|---|---| +| `acp-snapshot/` | ACP snapshot suite kit: subprocess scenario harness + golden normalizers + the `defineAcpSnapshotSuite` factory | (library — imported by example `*.snapshot.ts` suites) | | `invariants/` | Dev-mode event-contract invariants + session-log freeze | (listens on `session/*`, `agent/*`) | | `llm-replay/` | Record/replay adapter: short-circuits `llm/stream` from a recorded session JSONL (keyless snapshot tests) | (listens on `llm/stream`) | | `subagent-mock/` | Scripted `SubagentProvider` for deterministic seam/tool tests | (registers on `ctx.subagents`) | -`invariants` runs only in dev mode (contract checks, not runtime behavior). `llm-replay` backs the demos and the snapshot test tier under the per-file coverage gate. `subagent-mock` exercises the real `ctx.subagents` load path without a model or child agent. A package graduates OUT of `support/` into a product group only when it gains documented product consumers. +`invariants` runs only in dev mode (contract checks, not runtime behavior). `llm-replay` backs the demos and the snapshot test tier under the per-file coverage gate. `acp-snapshot` carries the snapshot tier's harness/normalizer/suite machinery so every example's suite is a scenario table over one shared, gate-covered implementation. `subagent-mock` exercises the real `ctx.subagents` load path without a model or child agent. A package graduates OUT of `support/` into a product group only when it gains documented product consumers. diff --git a/packages/support/acp-snapshot/README.md b/packages/support/acp-snapshot/README.md new file mode 100644 index 0000000000..324cfa9f1d --- /dev/null +++ b/packages/support/acp-snapshot/README.md @@ -0,0 +1,36 @@ +# `@deepseek-ai/dsh-acp-snapshot` + +The ACP snapshot suite kit: the shared machinery behind the keyless snapshot tier (`pnpm run test:snapshot`, [testing policy](../../../docs/testing.md)). An example gets a full snapshot suite from a scenario table plus a fixtures directory; every compare/guard mechanic lives here, under the per-file coverage gate, instead of being copied per example. + +Three layers, importable separately: + +- **`runScenario` (harness)** — boots the real agent bin as a subprocess via tsx (unbuilt, Loader path), drives it over ACP JSON-RPC stdio from a deterministic `input.json` script, tees raw stdout for the golden + purity check, and harvests every persisted session JSONL (parent + subagent children, primary-first) after a graceful stdin-EOF shutdown. Parameterized by `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath` — absolute paths; the subprocess cwd is a temp dir outside the repo). +- **Normalizers** — pure functions turning the two captured surfaces into stable text: `normalizeStdout` (JSON-RPC ids → first-seen sequence; UUIDs/cwd → tokens; doubles as the stdout-purity check), `normalizeSessionLog` (times zeroed, `seq` kept), and the composable `scrubRequestHeaders` (header bulk → `{{system}}`/`{{tools}}`, structure kept — [pinned-header RFC](../../../docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)). +- **`defineAcpSnapshotSuite` (factory)** — registers the whole describe/it tree for a scenario table: per-scenario golden + re-persisted-log compares, record-mode fixture write-back, the per-suite header pin with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin, non-pinning fixtures header-scrubbed). Must be called at vitest collection time. + +A consuming `*.snapshot.ts` is the scenario table plus one factory call: + +```ts +import { dirname, join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { defineAcpSnapshotSuite, type Scenario } from '@deepseek-ai/dsh-acp-snapshot' + +const SCENARIOS: Scenario[] = [ + { name: 'text-turn', hasModelTurn: true, recorded: true, pinsHeader: true }, +] + +defineAcpSnapshotSuite({ + agent: { // absolute paths, resolved from the suite's own location + binScript: fileURLToPath(new URL('../../../packages/ui/acp-agent/src/bin.ts', import.meta.url)), + configPath: fileURLToPath(new URL('../cordis.yml', import.meta.url)), + tsconfigPath: fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)), + }, + snapshotsDir: join(dirname(fileURLToPath(import.meta.url)), 'snapshots'), + scenarios: SCENARIOS, // exactly one entry sets pinsHeader + mode: process.env.DSH_SNAPSHOT === 'record' ? 'record' : 'replay', +}) +``` + +The example also ships a `cordis.snapshot.yml` replay overlay next to its `cordis.yml` (the bin swaps them under `DSH_SNAPSHOT=replay` — [single-source replay config RFC](../../../docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md)); replay fixtures are served by [`dsh-llm-replay`](../llm-replay/README.md), which this package points at via the `DSH_SNAPSHOT_*` env vars it sets on the child. Fixture roles, record/replay semantics, and scenario-table fields are documented on `Scenario` and in the [snapshot RFC](../../../docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md). + +Constraints: `suite.ts` imports vitest, so the package is importable only inside a vitest run (the harness and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the harness speaks the SDK's `ClientSideConnection` and answers `requestPermission` with `cancelled`. diff --git a/packages/support/acp-snapshot/package.json b/packages/support/acp-snapshot/package.json new file mode 100644 index 0000000000..363bc86e25 --- /dev/null +++ b/packages/support/acp-snapshot/package.json @@ -0,0 +1,35 @@ +{ + "name": "@deepseek-ai/dsh-acp-snapshot", + "description": "ACP snapshot suite kit: real-subprocess scenario harness, golden normalizers, and the suite factory behind the keyless snapshot tier", + "version": "0.0.1", + "private": true, + "type": "module", + "main": "lib/index.js", + "types": "lib/types/index.d.ts", + "exports": { + ".": { + "types": "./lib/types/index.d.ts", + "default": "./lib/index.js" + }, + "./src/*": "./src/*", + "./package.json": "./package.json" + }, + "files": [ + "lib/index.js", + "lib/types/**/*.d.ts", + "lib/types/**/*.d.ts.map", + "src" + ], + "license": "BSD-3-Clause", + "dependencies": { + "@agentclientprotocol/sdk": "0.25.1", + "tsx": "^4.22.4", + "vitest": "^4.1.8" + }, + "peerDependencies": { + "cordis": "^4.0.0-rc.6" + }, + "devDependencies": { + "cordis": "^4.0.0-rc.6" + } +} diff --git a/examples/acp-agent/tests/snapshot-harness.ts b/packages/support/acp-snapshot/src/harness.ts similarity index 86% rename from examples/acp-agent/tests/snapshot-harness.ts rename to packages/support/acp-snapshot/src/harness.ts index 8285b870bf..a46f752d5c 100644 --- a/examples/acp-agent/tests/snapshot-harness.ts +++ b/packages/support/acp-snapshot/src/harness.ts @@ -1,16 +1,19 @@ /** - * Shared harness for the ACP snapshot tests. A plain module (NOT a *.spec.ts / - * *.snapshot.ts) so importing it never re-registers another file's tests. + * Shared subprocess harness for ACP snapshot suites. A library module driven by + * the suite factory in ./suite.ts (and directly by harness-level specs); each + * example's `*.snapshot.ts` names its own agent-under-test paths. * - * It boots the REAL examples/acp-agent subprocess via the cordis Loader (so the + * It boots the REAL agent bin subprocess via the cordis Loader (so the * export-shape bug class stays guarded — see docs/postmortem/0001), drives it * over real ACP JSON-RPC stdio with a deterministic input script, tees raw * stdout (for the golden + a purity check) into an SDK `ClientSideConnection`, * and — in record mode — harvests the persisted session JSONL after a graceful - * shutdown flush. Two pure normalizers turn the captured stdout frames and the - * session-log events into stable, snapshot-able text. + * shutdown flush. The pure normalizers in ./normalize.ts turn the captured + * stdout frames and the session-log events into stable, snapshot-able text. * * See docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. + * + * @module @deepseek-ai/dsh-acp-snapshot/harness */ import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process' @@ -31,19 +34,36 @@ import { type SessionNotification, } from '@agentclientprotocol/sdk' -// The dsh-acp-agent bin (the demo:acp entry) and this example's cordis.yml. -// The bin resolves its config-path arg from CWD and, under DSH_SNAPSHOT=replay, -// swaps it for the sibling cordis.snapshot.yml. The child's cwd is a temp dir -// OUTSIDE the repo, so pass the example config's ABSOLUTE path. -const binScript = fileURLToPath(new URL('../../../packages/ui/acp-agent/src/bin.ts', import.meta.url)) -const configPath = fileURLToPath(new URL('../cordis.yml', import.meta.url)) +// Resolve tsx's ESM loader to an ABSOLUTE path once: the child runs with its +// cwd in a temp dir OUTSIDE the repo, where a bare `--import tsx` would not +// resolve from node_modules. import.meta.resolve gives this package's tsx +// regardless of the child cwd. const tsxLoader = fileURLToPath(import.meta.resolve('tsx')) -// The repo-root tsconfig: dev/test run UNBUILT and the `@deepseek-ai/dsh-*` -// imports resolve through its `paths` map. The child's cwd is a temp dir -// OUTSIDE the repo, so tsx's upward search would miss it — point tsx at the -// repo tsconfig explicitly (same fix the e2e harness uses). Repo root is four -// levels up from this file (examples/acp-agent/tests). -const repoTsconfig = fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)) + +/** + * The agent composition a scenario runs against: which bin to boot and which + * leaf config it loads. All paths are ABSOLUTE — the subprocess cwd is a temp + * dir outside the repo, so relative resolution would miss; a suite resolves + * them from its own `import.meta.url`. + */ +export interface AgentUnderTest { + /** The agent bin entry (e.g. `packages/ui/acp-agent/src/bin.ts`), run unbuilt via tsx. */ + binScript: string + /** + * The example's live `cordis.yml`. Under `DSH_SNAPSHOT=replay` the bin swaps + * it for the sibling `cordis.snapshot.yml` (the keyless replay overlay), so + * one path serves both modes. + */ + configPath: string + /** + * The repo-root tsconfig whose `paths` map resolves the unbuilt workspace + * imports. Passed to the child as `TSX_TSCONFIG_PATH`: tsx finds a tsconfig + * by searching UP from the child's cwd — a temp dir outside the repo — so + * without the explicit pin the dsh-* imports fail before the bin writes a + * byte. + */ + tsconfigPath: string +} /** * One step of a scenario's deterministic input script (`input.json`). The @@ -57,7 +77,7 @@ const repoTsconfig = fileURLToPath(new URL('../../../tsconfig.json', import.meta * the only way to exercise a cancel deterministically (a plain `prompt` step * awaits the response, which a cancel/hang scenario would block on forever). */ -type InputStep = +export type InputStep = | { op: 'initialize'; terminalOutput?: boolean } | { op: 'newSession' } | { op: 'newSessionExpectError'; additionalDirectories?: string[] } @@ -102,7 +122,10 @@ export interface RunResult { sessionLogs: HarvestedLog[] } -interface RunOptions { +/** How to run one scenario: the agent to boot, the mode, and the fixture wiring. */ +export interface RunOptions { + /** The agent composition to boot. */ + agent: AgentUnderTest /** `replay` (default, keyless) or `record` (real API, harvests the log). */ mode: 'replay' | 'record' /** The recorded session JSONL fixture path (replay reads it; record writes near it). */ @@ -130,6 +153,10 @@ interface RunOptions { * Run a scenario end-to-end against a freshly-spawned subprocess. Owns the * child and its temp dirs; always tears them down. Returns the captured stdout * and (record mode) the harvested session-log path. + * + * @param input The scenario's input script (steps + optional permission answers). + * @param opts The agent to boot, the mode, and the fixture wiring. + * @returns The captured stdout/stderr, session id, temp cwd, and harvested logs. */ export async function runScenario(input: InputScript, opts: RunOptions): Promise { const cwd = await mkdtemp(join(tmpdir(), 'acp-snap-cwd-')) @@ -151,7 +178,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise } const env: NodeJS.ProcessEnv = { ...process.env, - TSX_TSCONFIG_PATH: repoTsconfig, + TSX_TSCONFIG_PATH: opts.agent.tsconfigPath, DSH_SNAPSHOT: opts.mode, DSH_SNAPSHOT_FILE: opts.fixtureFile, DSH_SNAPSHOT_SESSIONS_ROOT: sessionsRoot, @@ -163,7 +190,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise child = spawn( process.execPath, - ['--import', tsxLoader, binScript, configPath], + ['--import', tsxLoader, opts.agent.binScript, opts.agent.configPath], { cwd, env, stdio: ['pipe', 'pipe', 'pipe'] }, ) @@ -302,9 +329,9 @@ async function runStep( // its own). To pin frame order deterministically, wait until the client // has OBSERVED the hang's streamed agent_message_chunk before cancelling — // so those update frames always precede the cancelled prompt response in - // the transcript (without this, the late chunk and the response race; see - // the Codex review of commit 5). Then cancel and await the prompt, which - // the bridge settles as `cancelled` once the abort propagates. + // the transcript (without this, the late chunk and the response race). + // Then cancel and await the prompt, which the bridge settles as + // `cancelled` once the abort propagates. const promptDone = client.prompt({ sessionId, prompt: [{ type: 'text', text: step.text }] }) await waitForUpdate(u => u.sessionUpdate === 'agent_message_chunk') await client.cancel({ sessionId }) @@ -335,8 +362,8 @@ function waitForExit(child: ChildProcessWithoutNullStreams): Promise { * * The JSONL backend lays sessions out as `//.jsonl` * (one bucket per cwd), so a parent and its same-cwd in-process child land in - * the SAME bucket — collecting all files across all buckets catches both (the - * old first-match short-circuit silently dropped the child). Returns `[]` if no + * the SAME bucket — collecting all files across all buckets catches both (a + * first-match short-circuit would silently drop the child). Returns `[]` if no * log was produced (a no-session scenario). */ async function harvestSessionLogs(root: string): Promise { diff --git a/packages/support/acp-snapshot/src/index.ts b/packages/support/acp-snapshot/src/index.ts new file mode 100644 index 0000000000..a0d3380086 --- /dev/null +++ b/packages/support/acp-snapshot/src/index.ts @@ -0,0 +1,37 @@ +/** + * ACP snapshot suite kit — the shared machinery behind the keyless snapshot + * tier (`pnpm run test:snapshot`). Three layers, composable per example: + * the subprocess scenario harness ({@link runScenario}), the pure golden + * normalizers ({@link normalizeStdout} / {@link normalizeSessionLog} / + * {@link scrubRequestHeaders}), and the suite factory + * ({@link defineAcpSnapshotSuite}) that registers a scenario table as a full + * describe/it tree. An example's `*.snapshot.ts` supplies only its + * {@link AgentUnderTest} paths, its snapshots directory, and its + * {@link Scenario} table. + * + * NOTE: ./suite.ts imports vitest, so this package is importable only inside a + * vitest run — a support-tier constraint stated in the README. + * + * @module @deepseek-ai/dsh-acp-snapshot + */ + +export { + runScenario, + type AgentUnderTest, + type HarvestedLog, + type InputScript, + type InputStep, + type RunOptions, + type RunResult, +} from './harness.ts' +export { + normalizeSessionLog, + normalizeStdout, + scrubRequestHeaders, + type NormalizeContext, +} from './normalize.ts' +export { + defineAcpSnapshotSuite, + type Scenario, + type SnapshotSuiteOptions, +} from './suite.ts' diff --git a/examples/acp-agent/tests/snapshot-normalize.ts b/packages/support/acp-snapshot/src/normalize.ts similarity index 91% rename from examples/acp-agent/tests/snapshot-normalize.ts rename to packages/support/acp-snapshot/src/normalize.ts index 28c10f102d..2cbe914b42 100644 --- a/examples/acp-agent/tests/snapshot-normalize.ts +++ b/packages/support/acp-snapshot/src/normalize.ts @@ -15,12 +15,15 @@ * A separate, composable normalizer — {@link scrubRequestHeaders} — replaces * the bulky request-header CONTENT (the composed system prompt and the tool * schema list) with `{{system}}`/`{{tools}}` tokens. It is deliberately NOT - * folded into {@link normalizeSessionLog}: the one header-pinning scenario - * compares that content verbatim, every other scenario composes the scrub in - * (the `pinsHeader` flag in acp.snapshot.ts; see the pinned-header RFC, + * folded into {@link normalizeSessionLog}: each suite's one header-pinning + * scenario compares that content verbatim, every other scenario composes the + * scrub in (the `pinsHeader` flag on the scenario table, consumed by the suite + * factory in ./suite.ts; see the pinned-header RFC, * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). * * See docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. + * + * @module @deepseek-ai/dsh-acp-snapshot/normalize */ const SESSION_ID = '{{sessionId}}' @@ -69,6 +72,10 @@ function scrubValue(value: unknown, ctx: NormalizeContext): unknown { * (1, 2, 3, …) and all volatile strings scrubbed. Throws if any non-empty line * is not valid JSON — that doubles as the stdout-purity check (no logger leaked * onto the protocol). + * + * @param rawStdout The captured stdout bytes, decoded utf8. + * @param ctx The run's volatile values to scrub. + * @returns The normalized NDJSON transcript, one frame per line. */ export function normalizeStdout(rawStdout: string, ctx: NormalizeContext): string { const lines = rawStdout.split('\n').filter(line => line.trim().length > 0) @@ -97,6 +104,10 @@ export function normalizeStdout(rawStdout: string, ctx: NormalizeContext): strin * zeroed/scrubbed, all volatile strings scrubbed, and `seq` is LEFT INTACT * (deterministic by contract). Output is JSONL in the same shape as the input — * one compact record per line. + * + * @param rawLog The raw session `.jsonl` content. + * @param ctx The run's volatile values to scrub. + * @returns The normalized JSONL log, one record per line. */ export function normalizeSessionLog(rawLog: string, ctx: NormalizeContext): string { const lines = rawLog.split('\n').filter(line => line.trim().length > 0) @@ -140,7 +151,10 @@ export function normalizeSessionLog(rawLog: string, ctx: NormalizeContext): stri * Only lines with something to scrub are re-serialized; every other line * passes through byte-for-byte, so the transform is idempotent and applying * it to an already-scrubbed fixture is a no-op — the on-disk-fixtures guard - * in acp.snapshot.ts relies on exactly that. + * in ./suite.ts relies on exactly that. + * + * @param rawLog The raw session `.jsonl` content. + * @returns The JSONL with header content tokenized, other lines byte-identical. */ export function scrubRequestHeaders(rawLog: string): string { const lines = rawLog.split('\n') diff --git a/packages/support/acp-snapshot/src/suite.ts b/packages/support/acp-snapshot/src/suite.ts new file mode 100644 index 0000000000..48d6692258 --- /dev/null +++ b/packages/support/acp-snapshot/src/suite.ts @@ -0,0 +1,355 @@ +/** + * The ACP snapshot suite factory (REPLAY by default, keyless). A suite is a + * scenario table plus a snapshots directory: each scenario under + * `//` ships an `input.json` (the client stdin script) and + * a `session.jsonl` fixture; replay boots the real agent subprocess + * (./harness.ts), drives it, and diffs the normalized stdout transcript + * against the committed `stdout.golden.jsonl`. For model scenarios it ALSO + * checks the re-persisted session log — against the `session.jsonl` fixture + * itself, not a separate golden: the fixture doubles as the replay source + * (recorded scenarios) and the expected produced log (both sides normalized + * before comparing). + * + * Request-header content (the composed system prompt + tool schemas riding on + * `request/header` events) is pinned by exactly ONE scenario per suite — the + * one with `pinsHeader` — and scrubbed to `{{system}}`/`{{tools}}` tokens in + * every other fixture and compare, so a prompt or tool-schema edit churns one + * committed line instead of every fixture. A per-run uniformity guard keeps + * the single pin sound: every live header must equal the pinned one, and no + * header-delta may appear outside the pinning scenario (see the + * pinned-header RFC, + * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). + * + * `pnpm run test:snapshot:record` (DSH_SNAPSHOT=record + -u) re-records the + * `session.jsonl` fixtures against the real API and refreshes the stdout golden + * in one pass; the caller resolves that env into {@link SnapshotSuiteOptions} + * (env reading stays at the suite edge, not in this library). + * + * @module @deepseek-ai/dsh-acp-snapshot/suite + */ + +import { readFile, readdir, writeFile } from 'node:fs/promises' +import { existsSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { type AgentUnderTest, type HarvestedLog, type InputScript, runScenario } from './harness.ts' +import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from './normalize.ts' + +/** A snapshot scenario and how its fixtures are produced. */ +export interface Scenario { + name: string + /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ + hasModelTurn: boolean + /** + * Whether the run persists a comparable session log to diff against the + * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn + * always produces a log worth comparing). Set it independently for a scenario + * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked + * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` + * events but never calls the model. + */ + comparesLog?: boolean + /** + * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` + * from the LIVE API. `recorded` scenarios are model-driven and reproducible; + * `authored` scenarios (a hand-written `replay.override.json` sidecar drives + * replay — e.g. a provider error or a cancel, which the live API can't be + * coaxed into deterministically — or a deterministic hook scenario whose + * derived empty script needs no sidecar) are NEVER re-recorded. + */ + recorded: boolean + /** + * How many SUBAGENT child sessions this scenario records beyond the top-level + * one (0 for a single-session scenario). Each child rides in a sibling fixture + * `session..jsonl` (1-based); replay forwards them to `dsh-llm-replay` so + * each child session replays from its own script, and record mode writes the + * harvested child logs back to those files. Defaults to 0. + */ + childSessions?: number + /** + * Whether THIS scenario's fixtures keep the full request-header content (the + * composed system prompt and tool schema list on `request/header` / + * `request/header-delta` events) and compare it verbatim. Exactly one + * scenario per suite pins it; every other scenario stores and compares that + * content as `{{system}}`/`{{tools}}` tokens ({@link scrubRequestHeaders}), + * so a system prompt or tool-schema change shows up as ONE committed-fixture + * diff, not one per scenario. One pin suffices because header composition is + * suite-uniform (parent, spawn child, and fork child all compose the same + * prompt-modulo-cwd and the same tools) — and that premise is ASSERTED, not + * assumed: every non-pinning run's live headers must equal the pinned + * fixture's (normalized), so a session-dependent header (say, a restricted + * subagent toolset) fails loud until it gets its own pinning scenario. + * Defaults to false. + */ + pinsHeader?: boolean +} + +/** One suite's inputs: the agent to boot, where its fixtures live, and its scenario table. */ +export interface SnapshotSuiteOptions { + /** The agent composition every scenario boots. */ + agent: AgentUnderTest + /** Absolute path of the suite's `snapshots/` directory (one subdir per scenario). */ + snapshotsDir: string + /** The scenario table; exactly one entry must set `pinsHeader`. */ + scenarios: Scenario[] + /** + * `replay` (keyless, the default tier) or `record` (live API; re-records the + * `recorded` scenarios' fixtures and refreshes the vitest goldens under + * `--update`). The caller derives this from `$DSH_SNAPSHOT` — env reading + * stays outside this library. + */ + mode: 'replay' | 'record' +} + +/** The sibling child-fixture paths for a scenario (`session.1.jsonl` …). */ +function childFixturePaths(dir: string, childSessions: number): string[] { + return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) +} + +/** + * Derive the {@link NormalizeContext} for a `session.jsonl` fixture from its own + * header line (`{ type: 'session', id, cwd }`). A committed fixture carries the + * session id and cwd of the run that harvested it — different from the live + * replay run — so normalizing it against the live run's ctx would leave those + * recorded values unscrubbed. Reading them from the header scrubs the fixture's + * own id/cwd to the same `{{sessionId}}`/`{{cwd}}` tokens the replay output gets. + * An authored fixture whose header is already normalized (`id:'{{sessionId}}'`, + * `cwd:'{{cwd}}'`) yields those tokens as the volatile values, so scrubbing them + * is an idempotent no-op. A header with no `cwd` falls back to a sentinel that + * cannot occur in a log (NOT `''`, which `String.split` would match on every + * character boundary and corrupt the output). + */ +function fixtureContext(fixture: string): NormalizeContext { + const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' + const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } + return { + sessionIds: typeof header.id === 'string' ? [header.id] : [], + cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', + } +} + +/** + * The `data.header` payload of every `request/header` event in a session + * JSONL, in log order, with the log's volatile values scrubbed first + * ({@link normalizeSessionLog}) so headers harvested from different runs — + * each embedding its own temp cwd in the composed prompt — compare on equal + * footing. + */ +function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { + return normalizeSessionLog(rawLog, ctx) + .split('\n') + .filter(line => line.trim().length > 0) + .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) + .filter(record => record.type === 'request/header') + .map(record => record.data?.header) +} + +/** Count the `request/header-delta` events in a session JSONL. */ +function headerDeltaCount(rawLog: string): number { + return rawLog.split('\n') + .filter(line => line.trim().length > 0) + .filter(line => (JSON.parse(line) as { type?: unknown }).type === 'request/header-delta') + .length +} + +/** + * Register the suite: one `describe` per scenario (the golden/log compares and + * the header-uniformity guard) plus the fixture guard block (no orphan + * scenario dirs, required files present, exactly one pin, non-pinning fixtures + * header-scrubbed). Must run at vitest collection time — it calls + * `describe`/`it`. Throws immediately if no scenario pins the header (the + * uniformity guard would have nothing to compare against). + * + * @param options The agent, snapshots directory, scenario table, and mode. + */ +export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void { + const { agent, snapshotsDir, scenarios, mode } = options + const RECORDING = mode === 'record' + + /** The suite's single header-pinning scenario. Guarded here (and by a meta-test) so the pin cannot silently vanish. */ + const pinningScenario = scenarios.find(s => s.pinsHeader === true) + if (pinningScenario === undefined) throw new Error('acp-snapshot: no scenario pins the request-header content') + + for (const scenario of scenarios) { + describe(`snapshot: ${scenario.name}`, () => { + // In RECORD mode, only re-run the `recorded` (live-API) scenarios; the + // `authored` ones (sidecar-driven errors/cancel) are never re-recorded. + it.skipIf(RECORDING && !scenario.recorded)('matches the goldens', async () => { + const dir = join(snapshotsDir, scenario.name) + const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript + const overrideFile = join(dir, 'replay.override.json') + const workspaceDir = join(dir, 'workspace') + const childSessions = scenario.childSessions ?? 0 + const result = await runScenario(input, { + agent, + mode, + fixtureFile: join(dir, 'session.jsonl'), + ...existsSync(overrideFile) ? { overrideFile } : {}, + // In REPLAY, forward the recorded child fixtures so each subagent session + // replays from its own script. In RECORD they are harvested, not read. + ...!RECORDING && childSessions > 0 ? { childFiles: childFixturePaths(dir, childSessions) } : {}, + ...existsSync(workspaceDir) ? { workspaceDir } : {}, + }) + + // Scrub every volatile id the run produced: the ACP server-issued session + // id plus every harvested log's recorded id (a subagent child id never + // surfaces over ACP, but it appears in the child's own log header). The + // normalizer's UUID catch-all covers any we don't enumerate. + const ctx: NormalizeContext = { + sessionIds: [ + ...result.sessionId !== undefined ? [result.sessionId] : [], + ...result.sessionLogs.map(l => l.id), + ], + cwd: result.cwd, + } + + // RECORD mode (recorded model scenarios only): persist the freshly-harvested + // logs back to their fixtures — the primary to session.jsonl, each child to + // session..jsonl in harvest order. `--update` refreshes the Vitest + // goldens but NOT these fixtures, so write them here. A non-pinning + // scenario's fixtures are written header-scrubbed, so a re-record can + // never smuggle the full prompt/schema content back into every fixture. + const scrub = scenario.pinsHeader === true + ? (log: string): string => log + : scrubRequestHeaders + if (RECORDING && scenario.recorded && scenario.hasModelTurn) { + expect(result.sessionLogs.length, 'record produced no session log to harvest').toBeGreaterThan(0) + expect(result.sessionLogs.length, `expected ${childSessions + 1} session logs (parent + children)`) + .toBe(childSessions + 1) + await writeFile(join(dir, 'session.jsonl'), scrub((result.sessionLogs[0] as HarvestedLog).content)) + for (let i = 1; i < result.sessionLogs.length; i++) { + await writeFile(join(dir, `session.${i}.jsonl`), scrub((result.sessionLogs[i] as HarvestedLog).content)) + } + } + + await expect(normalizeStdout(result.rawStdout, ctx)) + .toMatchFileSnapshot(join(dir, 'stdout.golden.jsonl')) + + // A model turn always produces a log worth comparing; a hook scenario can + // produce one without a model turn (a `rejected` turn carrying `hook/*`). + const comparesLog = scenario.comparesLog ?? scenario.hasModelTurn + if (comparesLog) { + // The harvested logs (primary-first) must match their committed fixtures + // 1:1. Each side passes through normalizeSessionLog, scrubbed against ITS + // OWN volatile values — the live run's via `ctx`, the committed fixture's + // via its own header (a committed file cannot share the live run's ids). + // Unless this scenario pins the header, both sides ALSO pass through + // scrubRequestHeaders: the live log carries the real prompt/schemas, the + // fixture carries the `{{system}}`/`{{tools}}` tokens, and the scrub is + // idempotent — so the compare checks the header's presence, position, + // reason, and config, but not its bulk content (pinned once, in the + // `pinsHeader` scenario). + expect(result.sessionLogs.length, 'this scenario must persist a session log').toBe(childSessions + 1) + const fixtureFiles = ['session.jsonl', ...Array.from({ length: childSessions }, (_, i) => `session.${i + 1}.jsonl`)] + for (let i = 0; i < fixtureFiles.length; i++) { + const harvested = scrub((result.sessionLogs[i] as HarvestedLog).content) + const fixture = scrub(await readFile(join(dir, fixtureFiles[i] as string), 'utf8')) + expect(normalizeSessionLog(harvested, ctx), `${fixtureFiles[i]} mismatch`) + .toEqual(normalizeSessionLog(fixture, fixtureContext(fixture))) + } + } + + // Header-uniformity guard: the single pin is sound only while every + // session in the suite composes the SAME header and keeps it for the + // whole run. Assert both halves live. (1) Every request/header the run + // produced (parent, spawn child, fork child, initial or resume) must + // equal the pinned fixture's header after each side is normalized + // against its own volatile values. (2) No request/header-delta may + // appear at all — a mid-run header change diverges from the pin by + // construction, and its content would be invisible under the scrub. If + // either fails, either the header changed (update the pin: re-record or + // hand-edit the pinning scenario's fixture) or composition became + // session-dependent by design (give the divergent shape its own + // pinning scenario). + if (scenario.pinsHeader !== true) { + const pinnedFixture = await readFile(join(snapshotsDir, pinningScenario.name, 'session.jsonl'), 'utf8') + const pinned = normalizedHeaders(pinnedFixture, fixtureContext(pinnedFixture)) + expect(pinned.length, `the pinning fixture (${pinningScenario.name}) must carry exactly one request/header`) + .toBe(1) + for (const log of result.sessionLogs) { + expect(headerDeltaCount(log.content), `session ${log.id}: a request/header-delta in a non-pinning scenario`) + .toBe(0) + const headers = normalizedHeaders(log.content, ctx) + for (const [k, header] of headers.entries()) { + expect(header, `session ${log.id}: request/header #${k + 1} diverged from the pinned (${pinningScenario.name}) header`) + .toEqual(pinned[0]) + } + } + } + }) + }) + } + + describe('snapshot fixtures', () => { + it('every scenario directory is registered (no orphans)', async () => { + // toMatchFileSnapshot does not prune orphaned golden/fixture files, so a + // renamed/removed scenario could leave a stale dir that nothing exercises. + // Fail loud on any snapshots/ not present in the scenario table. + const entries = await readdir(snapshotsDir, { withFileTypes: true }) + const onDisk = entries.filter(e => e.isDirectory()).map(e => e.name).sort() + const registered = scenarios.map(s => s.name).sort() + expect(onDisk).toEqual(registered) + }) + + it('every registered scenario has its required fixture files', () => { + // Every scenario has an input script and an stdout golden. EVERY scenario + // also needs `session.jsonl`: the suite boots `llm-replay` with that path + // as the replay source for ALL scenarios (the factory passes + // `fixtureFile: /session.jsonl` unconditionally), and `loadReplayScript` + // throws "fixture not found" when it is absent and no override replaces it. + // A no-model scenario ships a header-only `session.jsonl` (it derives to an + // empty script — no model call is made); a model scenario's fixture also + // doubles as the expected-log artifact the run is diffed against. An authored + // (non-`recorded`) model scenario additionally ships a `replay.override.json` + // sidecar for the throw/hang cases a derived script cannot express. + for (const { name, hasModelTurn, recorded, childSessions } of scenarios) { + const dir = join(snapshotsDir, name) + expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true) + expect(existsSync(join(dir, 'stdout.golden.jsonl')), `${name}/stdout.golden.jsonl`).toBe(true) + expect(existsSync(join(dir, 'session.jsonl')), `${name}/session.jsonl`).toBe(true) + if (hasModelTurn && !recorded) { + expect(existsSync(join(dir, 'replay.override.json')), `${name}/replay.override.json`).toBe(true) + } + // A nested-agent scenario ships one child fixture per recorded subagent + // session (`session.1.jsonl` …), the replay source for that child session. + for (const childFixture of childFixturePaths(dir, childSessions ?? 0)) { + expect(existsSync(childFixture), childFixture).toBe(true) + } + } + }) + + it('exactly one scenario pins the request-header content', () => { + // Zero pins would drop the prompt/schema surface from the suite entirely; + // two would split it. One pin per suite is the design (pinned-header RFC); + // WHICH scenario pins is the scenario table's reviewable choice. + expect(scenarios.filter(s => s.pinsHeader === true).map(s => s.name)).toEqual([pinningScenario.name]) + }) + + it('committed fixtures carry request-header content ONLY in the pinning scenario', async () => { + // The whole point of the pin: a system-prompt or tool-schema change must + // churn exactly one committed line. A non-pinning fixture that carries the + // full header (a hand-recorded file, or a header line hand-edited out of + // its canonical JSON form) silently reopens the suite-wide churn, so fail + // loud here: every non-pinning session*.jsonl must be a fixed point of + // scrubRequestHeaders (apply the scrub to fix a violation), and the + // pinning scenario's fixtures must NOT be (their content IS the pin). + for (const scenario of scenarios) { + const dir = join(snapshotsDir, scenario.name) + const files = [ + 'session.jsonl', + ...Array.from({ length: scenario.childSessions ?? 0 }, (_, i) => `session.${i + 1}.jsonl`), + ] + for (const file of files) { + const fixture = await readFile(join(dir, file), 'utf8') + if (scenario.pinsHeader === true) { + expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} must PIN the full header content`) + .not.toEqual(fixture) + } else { + expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} carries unscrubbed header content`) + .toEqual(fixture) + } + } + } + }) + }) +} diff --git a/examples/acp-agent/tests/snapshot-normalize.spec.ts b/packages/support/acp-snapshot/tests/normalize.spec.ts similarity index 98% rename from examples/acp-agent/tests/snapshot-normalize.spec.ts rename to packages/support/acp-snapshot/tests/normalize.spec.ts index fa225bd659..56c15306a3 100644 --- a/examples/acp-agent/tests/snapshot-normalize.spec.ts +++ b/packages/support/acp-snapshot/tests/normalize.spec.ts @@ -1,9 +1,9 @@ import { describe, expect, it } from 'vitest' -import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from '../tests/snapshot-normalize.ts' +import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from '../src/normalize.ts' /** * Unit tests for the pure snapshot normalizers. Live as a *.spec.ts (runs in - * the default unit gate) and import the harness-side normalizers directly. + * the default unit gate) and import the normalizers directly. */ const ctx: NormalizeContext = { diff --git a/packages/support/acp-snapshot/tsconfig.json b/packages/support/acp-snapshot/tsconfig.json new file mode 100644 index 0000000000..749cb0208e --- /dev/null +++ b/packages/support/acp-snapshot/tsconfig.json @@ -0,0 +1,11 @@ +{ + "extends": "../../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "src", + "outDir": "lib/types" + }, + "include": [ + "src" + ], + "references": [] +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index b501bc21ee..9856e99e95 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -748,6 +748,22 @@ importers: specifier: ^4.0.0-rc.6 version: 4.0.0-rc.6(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.4) + packages/support/acp-snapshot: + dependencies: + '@agentclientprotocol/sdk': + specifier: 0.25.1 + version: 0.25.1(zod@4.4.3) + tsx: + specifier: ^4.22.4 + version: 4.22.4 + vitest: + specifier: ^4.1.8 + version: 4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + devDependencies: + cordis: + specifier: ^4.0.0-rc.6 + version: 4.0.0-rc.6(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.4) + packages/support/invariants: devDependencies: '@deepseek-ai/dsh-agent': @@ -5241,6 +5257,14 @@ snapshots: optionalDependencies: vite: 8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + '@vitest/mocker@4.1.8(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0))': + dependencies: + '@vitest/spy': 4.1.8 + estree-walker: 3.0.3 + magic-string: 0.30.21 + optionalDependencies: + vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + '@vitest/pretty-format@4.1.8': dependencies: tinyrainbow: 3.1.0 @@ -6891,6 +6915,21 @@ snapshots: tsx: 4.22.4 yaml: 2.9.0 + vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0): + dependencies: + lightningcss: 1.32.0 + picomatch: 4.0.4 + postcss: 8.5.15 + rolldown: 1.0.3 + tinyglobby: 0.2.17 + optionalDependencies: + '@types/node': 25.9.3 + esbuild: 0.28.1 + fsevents: 2.3.3 + jiti: 2.7.0 + tsx: 4.22.4 + yaml: 2.9.0 + vitest@4.1.8(@types/node@22.20.0)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): dependencies: '@vitest/expect': 4.1.8 @@ -6920,6 +6959,35 @@ snapshots: transitivePeerDependencies: - msw + vitest@4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): + dependencies: + '@vitest/expect': 4.1.8 + '@vitest/mocker': 4.1.8(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + '@vitest/pretty-format': 4.1.8 + '@vitest/runner': 4.1.8 + '@vitest/snapshot': 4.1.8 + '@vitest/spy': 4.1.8 + '@vitest/utils': 4.1.8 + es-module-lexer: 2.1.0 + expect-type: 1.3.0 + magic-string: 0.30.21 + obug: 2.1.3 + pathe: 2.0.3 + picomatch: 4.0.4 + std-env: 4.1.0 + tinybench: 2.9.0 + tinyexec: 1.2.4 + tinyglobby: 0.2.17 + tinyrainbow: 3.1.0 + vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + why-is-node-running: 2.3.0 + optionalDependencies: + '@types/node': 25.9.3 + '@vitest/coverage-v8': 4.1.8(vitest@4.1.8) + jsdom: 29.1.1 + transitivePeerDependencies: + - msw + w3c-xmlserializer@5.0.0: dependencies: xml-name-validator: 5.0.0 diff --git a/tsconfig.build.json b/tsconfig.build.json index b6ba7901f2..96d87c01e9 100644 --- a/tsconfig.build.json +++ b/tsconfig.build.json @@ -44,6 +44,7 @@ { "path": "./packages/ui/app-boot" }, { "path": "./packages/ui/stdio-agent" }, { "path": "./packages/support/llm-replay" }, + { "path": "./packages/support/acp-snapshot" }, { "path": "./packages/subagent/subagent" }, { "path": "./packages/support/subagent-mock" }, { "path": "./packages/subagent/tool-subagent" }, diff --git a/tsconfig.json b/tsconfig.json index 9cd7aa8a6d..ff737baf04 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -55,6 +55,7 @@ { "path": "./packages/ui/app-boot" }, { "path": "./packages/ui/stdio-agent" }, { "path": "./packages/support/llm-replay" }, + { "path": "./packages/support/acp-snapshot" }, { "path": "./packages/subagent/subagent" }, { "path": "./packages/support/subagent-mock" }, { "path": "./packages/subagent/tool-subagent" },