Merge branch 'master' into worktree-windows-runtime
This commit is contained in:
@@ -6,12 +6,12 @@
|
||||
* It boots the REAL agent bin subprocess via the cordis Loader (so the
|
||||
* export-shape bug class stays guarded — see docs/postmortem/0001), drives it
|
||||
* over real ACP JSON-RPC stdio with a deterministic input script, tees raw
|
||||
* stdout (for the golden + a purity check) into an SDK `ClientSideConnection`,
|
||||
* stdout (for the expected-output and purity checks) into an SDK `ClientSideConnection`,
|
||||
* and — in record mode — harvests the persisted session JSONL after a graceful
|
||||
* shutdown flush. The pure normalizers in ./normalize.ts turn the captured
|
||||
* stdout frames and the session-log events into stable, snapshot-able text.
|
||||
*
|
||||
* See docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md.
|
||||
* See .agents/notes/implemented/testing/2026-06-19-acp-snapshot-tests.md.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-acp-snapshot/harness
|
||||
*/
|
||||
@@ -171,7 +171,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise
|
||||
const cwd = await mkdtemp(join(tmpdir(), 'acp-snap-cwd-'))
|
||||
const sessionsRoot = await mkdtemp(join(tmpdir(), 'acp-snap-sessions-'))
|
||||
// Fixed path length: spill-policy budgets the preview against the REAL path
|
||||
// before stdout normalization, so tmpdir() length differences churn goldens.
|
||||
// before stdout normalization, so tmpdir() length differences churn expected outputs.
|
||||
const spillRoot = snapshotSpillRoot()
|
||||
// Everything past the temp-dir creation is followed by failure-safe cleanup,
|
||||
// so a failure in workspace seeding, spawn, or any step never leaks resources.
|
||||
@@ -180,7 +180,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise
|
||||
let sessionLogs: HarvestedLog[] = []
|
||||
const outcome = await (async (): Promise<RunResult> => {
|
||||
// Seed the workspace if the scenario ships one (a file the agent reads/edits).
|
||||
// Copied into the temp cwd so the agent's bash tools see it; the goldens
|
||||
// Copied into the temp cwd so the agent's bash tools see it; the expected outputs
|
||||
// normalize the cwd, so the seeded paths stay stable across runs.
|
||||
if (opts.workspaceDir !== undefined && existsSync(opts.workspaceDir)) {
|
||||
await cp(opts.workspaceDir, cwd, { recursive: true })
|
||||
@@ -404,11 +404,11 @@ async function runStep(
|
||||
* header line, and return them ordered primary-first: the top-level session (no
|
||||
* `parentSession`) leads, then each subagent child by ascending `createdAt`.
|
||||
*
|
||||
* The JSONL backend lays sessions out as `<root>/<cwd-bucket>/<encoded-id>.jsonl`
|
||||
* (one bucket per cwd), so a parent and its same-cwd in-process child land in
|
||||
* the SAME bucket — collecting all files across all buckets catches both (a
|
||||
* first-match short-circuit would silently drop the child). Returns `[]` if no
|
||||
* log was produced (a no-session scenario).
|
||||
* Snapshot configs select the JSONL backend's raw mode, which lays sessions
|
||||
* out as `<root>/<cwd-bucket>/<encoded-id>.jsonl` (one bucket per cwd). A
|
||||
* parent and its same-cwd in-process child land in the SAME bucket, so
|
||||
* collecting all files across all buckets catches both. Returns `[]` if no log
|
||||
* was produced (a no-session scenario).
|
||||
*/
|
||||
async function harvestSessionLogs(root: string): Promise<HarvestedLog[]> {
|
||||
let cwdDirs: string[]
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
* ACP snapshot suite kit — the shared machinery behind the keyless snapshot
|
||||
* tier (`pnpm run test:snapshot`). Four layers, composable per example: the
|
||||
* shared subprocess/client launcher ({@link launchAcpTestAgent}), the scripted
|
||||
* scenario harness ({@link runScenario}), the pure golden normalizers
|
||||
* scenario harness ({@link runScenario}), the pure expected-output normalizers
|
||||
* ({@link normalizeStdout} / {@link normalizeSessionLog} /
|
||||
* {@link scrubRequestHeaders} / {@link scrubSystemPrompts}), and the suite
|
||||
* factory ({@link defineAcpSnapshotSuite}) that registers a scenario table as a
|
||||
|
||||
@@ -92,7 +92,7 @@ function scrubValue(value: unknown, ctx: NormalizeContext, cwdPathMode: CwdPathM
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize a raw stdout transcript (newline-delimited JSON-RPC frames) into a stable golden
|
||||
* Normalize a raw stdout transcript (newline-delimited JSON-RPC frames) into a stable expected output
|
||||
* in the same shape as the wire: one compact JSON frame per line (NDJSON), with the JSON-RPC
|
||||
* `id` rewritten to a per-transcript sequence (1, 2, 3, …) and all volatile strings scrubbed.
|
||||
* Invalid JSON throws, doubling as a protocol-stdout purity check.
|
||||
@@ -110,7 +110,7 @@ export function normalizeStdout(
|
||||
const cwdPathMode = options.cwdPathMode ?? 'canonical'
|
||||
const lines = rawStdout.split('\n').filter(line => line.trim().length > 0)
|
||||
// Map each distinct JSON-RPC id (request/response correlate by id) to a stable
|
||||
// sequence number, in first-seen order, so id churn doesn't perturb the golden.
|
||||
// sequence number, in first-seen order, so id churn doesn't perturb the expected output.
|
||||
const idSeq = new Map<string, number>()
|
||||
const stableId = (id: unknown): number => {
|
||||
const key = JSON.stringify(id)
|
||||
@@ -129,7 +129,7 @@ export function normalizeStdout(
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize a session JSONL log into a stable golden: the header line's
|
||||
* Normalize a session JSONL log into a stable expected output: the header line's
|
||||
* volatile fields (`createdAt`, `id`, `cwd`) and every event's `time` are
|
||||
* zeroed/scrubbed, all volatile strings scrubbed, and `seq` is LEFT INTACT
|
||||
* (deterministic by contract). Output is JSONL in the same shape as the input —
|
||||
@@ -156,7 +156,7 @@ export function normalizeSessionLog(
|
||||
// Event line: zero the epoch-ms timestamp; keep seq (deterministic).
|
||||
record.time = 0
|
||||
// A hook/result carries the hook's wall-clock runtime (`data.durationMs`),
|
||||
// which is run-to-run noise like `time` — zero it so the golden reflects
|
||||
// which is run-to-run noise like `time` — zero it so the expected output reflects
|
||||
// the hook's decision/exit, not how long the shell took.
|
||||
if (record.type === 'hook/result' && record.data !== null && typeof record.data === 'object') {
|
||||
const data = record.data as Record<string, unknown>
|
||||
|
||||
@@ -31,13 +31,13 @@ import {
|
||||
} from './normalize.ts'
|
||||
|
||||
/** The readable system-prompt snapshot beside each header-pinning fixture. */
|
||||
const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.golden.md'
|
||||
const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.expected.md'
|
||||
|
||||
/** The structured tool-schema snapshot beside each header-pinning fixture. */
|
||||
const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.golden.json'
|
||||
const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.expected.json'
|
||||
|
||||
/** The optional full Windows-native stdout transcript. */
|
||||
const WINDOWS_STDOUT_SNAPSHOT = 'stdout.golden.windows.jsonl'
|
||||
const WINDOWS_STDOUT_SNAPSHOT = 'stdout.expected.windows.jsonl'
|
||||
|
||||
/** Stable session-log token standing in for the sidecar's initial schemas. */
|
||||
const TOOLS_TOKEN = '{{tools}}'
|
||||
@@ -45,7 +45,7 @@ const TOOLS_TOKEN = '{{tools}}'
|
||||
/** A snapshot scenario and how its fixtures are produced. */
|
||||
export interface Scenario {
|
||||
name: string
|
||||
/** Whether the scenario drives at least one model turn (so a JSONL golden applies). */
|
||||
/** Whether the scenario drives at least one model turn (so a JSONL expected output applies). */
|
||||
hasModelTurn: boolean
|
||||
/**
|
||||
* Whether the run persists a comparable session log to diff against the
|
||||
@@ -106,7 +106,7 @@ export interface Scenario {
|
||||
configPath?: string
|
||||
/**
|
||||
* Whether Windows additionally compares stdout with native separators against
|
||||
* `stdout.golden.windows.jsonl`. The shared canonical stdout golden is still
|
||||
* `stdout.expected.windows.jsonl`. The shared canonical stdout expected output is still
|
||||
* compared on every platform, and the fixture guard requires this sidecar
|
||||
* exactly when the option is set.
|
||||
*/
|
||||
@@ -139,24 +139,24 @@ export function scenarioSkipped(
|
||||
return scenario.posixOnly === true && platform === 'win32'
|
||||
}
|
||||
|
||||
/** One stdout golden selected for a platform run. */
|
||||
interface StdoutGoldenVariant {
|
||||
/** One stdout expected output selected for a platform run. */
|
||||
interface StdoutExpectedVariant {
|
||||
file: string
|
||||
cwdPathMode: CwdPathMode
|
||||
}
|
||||
|
||||
/**
|
||||
* Select the shared stdout golden plus any platform-native assertion declared by a scenario.
|
||||
* Select the shared stdout expected output plus any platform-native assertion declared by a scenario.
|
||||
*
|
||||
* @param scenario The scenario whose stdout contract is being selected.
|
||||
* @param platform The running Node platform, injectable for unit coverage.
|
||||
* @returns The ordered golden variants: shared canonical first, then optional Windows native.
|
||||
* @returns The ordered expected-output variants: shared canonical first, then optional Windows native.
|
||||
*/
|
||||
export function stdoutGoldenVariants(
|
||||
export function stdoutExpectedVariants(
|
||||
scenario: Scenario,
|
||||
platform: NodeJS.Platform = process.platform,
|
||||
): StdoutGoldenVariant[] {
|
||||
const canonical: StdoutGoldenVariant = { file: 'stdout.golden.jsonl', cwdPathMode: 'canonical' }
|
||||
): StdoutExpectedVariant[] {
|
||||
const canonical: StdoutExpectedVariant = { file: 'stdout.expected.jsonl', cwdPathMode: 'canonical' }
|
||||
if (platform !== 'win32' || scenario.pinsNativeWindowsStdout !== true) return [canonical]
|
||||
return [canonical, { file: WINDOWS_STDOUT_SNAPSHOT, cwdPathMode: 'native' }]
|
||||
}
|
||||
@@ -171,8 +171,8 @@ export interface SnapshotSuiteOptions {
|
||||
scenarios: Scenario[]
|
||||
/**
|
||||
* `replay` (keyless, the default tier), `record` (live API; re-records the
|
||||
* `recorded` scenarios' fixtures and refreshes the Vitest goldens under
|
||||
* `--update`), or `refresh` (keyless replay that rewrites stdout goldens and
|
||||
* `recorded` scenarios' fixtures and refreshes the Vitest expected outputs under
|
||||
* `--update`), or `refresh` (keyless replay that rewrites stdout expected outputs and
|
||||
* comparable session fixtures from the replay run). The caller derives this
|
||||
* from `$DSH_SNAPSHOT` — env reading stays outside this library.
|
||||
*/
|
||||
@@ -486,7 +486,7 @@ export function stabilizeRefreshLog(fresh: string, existing: string, replacement
|
||||
}
|
||||
|
||||
/**
|
||||
* Register the suite: one test per scenario (the golden/log compares and
|
||||
* Register the suite: one test per scenario (the expected-output and log comparisons and
|
||||
* the header-uniformity guard) plus the fixture guard block (no orphan
|
||||
* scenario dirs, required files present, exactly one pin per header class,
|
||||
* pinning fixtures well-formed, every JSONL prompt-scrubbed, non-pinning
|
||||
@@ -527,7 +527,7 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
|
||||
// In RECORD mode, only re-run the `recorded` (live-API) scenarios; the `authored` ones
|
||||
// (sidecar-driven errors/cancel) are never re-recorded. `posixOnly` scenarios skip on
|
||||
// Windows, where their process semantics cannot be driven.
|
||||
it.skipIf(scenarioSkipped(scenario, RECORDING))(`snapshot: ${scenario.name} matches the goldens`, async ({ expect }) => {
|
||||
it.skipIf(scenarioSkipped(scenario, RECORDING))(`snapshot: ${scenario.name} matches the expected outputs`, async ({ expect }) => {
|
||||
const dir = join(snapshotsDir, scenario.name)
|
||||
const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript
|
||||
const overrideFile = join(dir, 'replay.override.json')
|
||||
@@ -631,12 +631,12 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
|
||||
}
|
||||
}
|
||||
|
||||
for (const golden of stdoutGoldenVariants(scenario)) {
|
||||
const stdout = normalizeStdout(result.rawStdout, ctx, { cwdPathMode: golden.cwdPathMode })
|
||||
for (const expected of stdoutExpectedVariants(scenario)) {
|
||||
const stdout = normalizeStdout(result.rawStdout, ctx, { cwdPathMode: expected.cwdPathMode })
|
||||
if (REFRESHING) {
|
||||
await writeFile(join(dir, golden.file), stdout)
|
||||
await writeFile(join(dir, expected.file), stdout)
|
||||
}
|
||||
await expect(stdout, `${golden.file} mismatch`).toMatchFileSnapshot(join(dir, golden.file))
|
||||
await expect(stdout, `${expected.file} mismatch`).toMatchFileSnapshot(join(dir, expected.file))
|
||||
}
|
||||
|
||||
// A model turn always produces a log worth comparing; a hook scenario can
|
||||
@@ -713,7 +713,7 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
|
||||
|
||||
describe('snapshot fixtures', () => {
|
||||
it('every scenario directory is registered (no orphans)', async () => {
|
||||
// toMatchFileSnapshot does not prune orphaned golden/fixture files, so a
|
||||
// toMatchFileSnapshot does not prune orphaned expected-output or fixture files, so a
|
||||
// renamed/removed scenario could leave a stale dir that nothing exercises.
|
||||
// Fail loud on any snapshots/<dir> not present in the scenario table.
|
||||
const entries = await readdir(snapshotsDir, { withFileTypes: true })
|
||||
@@ -727,7 +727,7 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
|
||||
for (const { name, overridden, pinsHeader, pinsNativeWindowsStdout } of scenarios) {
|
||||
const dir = join(snapshotsDir, name)
|
||||
expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true)
|
||||
expect(existsSync(join(dir, 'stdout.golden.jsonl')), `${name}/stdout.golden.jsonl`).toBe(true)
|
||||
expect(existsSync(join(dir, 'stdout.expected.jsonl')), `${name}/stdout.expected.jsonl`).toBe(true)
|
||||
expect(
|
||||
existsSync(join(dir, WINDOWS_STDOUT_SNAPSHOT)),
|
||||
`${name}/${WINDOWS_STDOUT_SNAPSHOT} presence must match \`pinsNativeWindowsStdout\``,
|
||||
|
||||
Reference in New Issue
Block a user