/** * Keyless-by-default ACP snapshot suite factory. Each scenario drives the real * subprocess and compares normalized stdout; comparable session fixtures are * both replay input and expected output. Record mode refreshes reproducible * model scenarios from the live API, while refresh mode replays committed * scripts and rewrites derived artifacts without a key. * Replay scenarios run concurrently because each subprocess owns unique temp * cwd and persistence roots and reads only committed fixtures. Record and * refresh stay serial while writing. * * Exactly one scenario per header-composition class pins the full prompt and * tool-schema sequences in dedicated sidecars. Every live header is checked * against that pin, so session-dependent composition must declare a separate * class instead of escaping coverage. * @module @deepseek-ai/dsh-acp-snapshot/suite */ import { readFile, readdir, rm, writeFile } from 'node:fs/promises' import { existsSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { type AgentUnderTest, type HarvestedLog, type InputScript, runScenario } from './harness.ts' import { type CwdPathMode, type NormalizeContext, extractSnapshotSpillPaths, normalizeSessionLog, normalizeStdout, scrubRequestHeaders, scrubSystemPrompts, scrubToolSchemas, } from './normalize.ts' /** The readable system-prompt snapshot beside each header-pinning fixture. */ const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.expected.md' /** The structured tool-schema snapshot beside each header-pinning fixture. */ const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.expected.json' /** The optional full Windows-native stdout transcript. */ const WINDOWS_STDOUT_SNAPSHOT = 'stdout.expected.windows.jsonl' /** Stable session-log token standing in for the sidecar's initial schemas. */ const TOOLS_TOKEN = '{{tools}}' const PACKED_CHUNK_ROW_TYPES = new Set(['text-chunks', 'reasoning-chunks', 'tool-call-chunks']) /** A snapshot scenario and how its fixtures are produced. */ export interface Scenario { name: string /** Deployment environment for this scenario's subprocess. */ env?: NodeJS.ProcessEnv /** Whether the scenario drives at least one model turn (so a JSONL expected output applies). */ hasModelTurn: boolean /** * Whether the run persists a comparable session log to diff against the * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn * always produces a log worth comparing). Set it independently for a scenario * that produces a non-trivial durable log without calling the model. */ comparesLog?: boolean /** * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` * from the LIVE API. `recorded` scenarios are model-driven and reproducible; * `authored` scenarios (fixtures hand-written or hand-harvested — e.g. a * provider error or a cancel the live API can't be coaxed into * deterministically, a deterministic hook scenario, or a scripted repetition * a live model won't reproduce) are NEVER re-recorded. */ recorded: boolean /** * Whether replay is driven by a hand-written `replay.override.json` sidecar * (a `ReplayOverrideDoc` that replaces or patches the script derived from * `session.jsonl`) — the throw/hang cases chunks cannot express. The fixture * guard requires the sidecar exactly when this is set: the harness forwards * the file purely on existence, so an unregistered stray sidecar would * silently alter the derived script. The guard fails loud on either * mismatch. Defaults to false (replay derives from the fixture's * `assistant/chunk` events). */ overridden?: boolean /** * Whether this scenario is its header class's sole request-header pin. Dedicated sidecars own * the prompt and tool schemas, while every classmate is checked for equality. */ pinsHeader?: boolean /** * How many changed `request/header` snapshots this PINNING scenario's primary * fixture legitimately carries (default 0). Their full prompt text is kept in * the readable Markdown pin; any other count fails. Meaningless off the pin. */ expectedHeaderChanges?: number /** * Which header-composition class this scenario belongs to. Scenarios that * boot the same config compose the same header; each class has exactly one * {@link pinsHeader} scenario, and the uniformity guard compares every * other member against ITS class's pin. Defaults to `'default'`; a * scenario booting an alternate config ({@link configPath}) whose tool * list or prompt sections differ by construction carries its own class. */ headerClass?: string /** * Alternate LIVE config path (absolute) this scenario boots instead of * {@link AgentUnderTest.configPath} — an overlay composing a different * tree (its basename must still end in `cordis.yml` so the bin's replay * swap finds the sibling `*cordis.snapshot.yml`). A scenario whose * overlay changes the composed header also needs its own * {@link headerClass}. */ configPath?: string /** * Parent directory for the generated session cwd. Defaults to the platform * temp directory; set this when temp is itself part of the behavior under * test and the scenario needs an independent project location. */ workspaceParent?: string /** * Whether Windows additionally compares stdout with native separators against * `stdout.expected.windows.jsonl`. The shared canonical stdout expected output is still * compared on every platform, and the fixture guard requires this sidecar * exactly when the option is set. */ pinsNativeWindowsStdout?: boolean /** * Whether the driven behavior needs POSIX process semantics the harness * cannot exercise on Windows (e.g. cancelling a live bash tool call kills a * detached process group). The scenario's run test is skipped on Windows; * its fixtures stay guarded on every platform. */ posixOnly?: boolean } /** * Whether a scenario's run test is skipped for this mode and host: record mode * skips authored (non-`recorded`) scenarios, and {@link Scenario.posixOnly} * scenarios skip on Windows. * * @param scenario The scenario whose run test is being registered. * @param recording Whether the suite runs in record mode. * @param platform The running Node platform, injectable for unit coverage. * @returns True when the scenario's run test must not execute. */ export function scenarioSkipped( scenario: Scenario, recording: boolean, platform: NodeJS.Platform = process.platform, ): boolean { if (recording && !scenario.recorded) return true return scenario.posixOnly === true && platform === 'win32' } /** One stdout expected output selected for a platform run. */ interface StdoutExpectedVariant { file: string cwdPathMode: CwdPathMode } /** * Select the shared stdout expected output plus any platform-native assertion declared by a scenario. * * @param scenario The scenario whose stdout contract is being selected. * @param platform The running Node platform, injectable for unit coverage. * @returns The ordered expected-output variants: shared canonical first, then optional Windows native. */ export function stdoutExpectedVariants( scenario: Scenario, platform: NodeJS.Platform = process.platform, ): StdoutExpectedVariant[] { const canonical: StdoutExpectedVariant = { file: 'stdout.expected.jsonl', cwdPathMode: 'canonical' } if (platform !== 'win32' || scenario.pinsNativeWindowsStdout !== true) return [canonical] return [canonical, { file: WINDOWS_STDOUT_SNAPSHOT, cwdPathMode: 'native' }] } /** One suite's inputs: the agent to boot, where its fixtures live, and its scenario table. */ export interface SnapshotSuiteOptions { /** The agent composition every scenario boots. */ agent: AgentUnderTest /** Absolute path of the suite's `snapshots/` directory (one subdir per scenario). */ snapshotsDir: string /** The scenario table; exactly one entry per header class must set `pinsHeader`. */ scenarios: Scenario[] /** * `replay` (keyless, the default tier), `record` (live API; re-records the * `recorded` scenarios' fixtures and refreshes the Vitest expected outputs under * `--update`), or `refresh` (keyless replay that rewrites stdout expected outputs and * comparable session fixtures from the replay run). The caller derives this * from `$DSH_SNAPSHOT` — env reading stays outside this library. */ mode: 'replay' | 'record' | 'refresh' } /** * Validate and order a scenario directory's session-fixture filenames. * * The primary fixture is always `session.jsonl`; child sessions are discovered * from contiguous `session.1.jsonl` … filenames. The directory is the source of * truth, so scenario tables do not duplicate a child count that can drift from * the files. A session-like JSONL with any other suffix fails loud. * * @param names File names in one scenario directory. * @returns The primary and child fixture names in replay/harvest order. */ export function sessionFixtureNames(names: readonly string[]): string[] { if (!names.includes('session.jsonl')) throw new Error('missing session.jsonl') const children: { name: string; index: number }[] = [] for (const name of names) { if (name === 'session.jsonl') continue if (!name.startsWith('session.') || !name.endsWith('.jsonl')) continue const match = /^session\.([1-9]\d*)\.jsonl$/.exec(name) if (match === null) throw new Error(`invalid child session fixture name: ${name}`) children.push({ name, index: Number(match[1]) }) } children.sort((a, b) => a.index - b.index) for (const [offset, child] of children.entries()) { const expected = offset + 1 if (child.index !== expected) { throw new Error(`child session fixtures must be contiguous: expected session.${expected}.jsonl, found ${child.name}`) } } return ['session.jsonl', ...children.map(child => child.name)] } /** Read one scenario directory's validated session-fixture inventory. */ async function sessionFixtures(dir: string): Promise { const entries = await readdir(dir, { withFileTypes: true }) return sessionFixtureNames(entries.filter(entry => entry.isFile()).map(entry => entry.name)) } /** * Derive normalization values from a fixture's own session header. Recorded ids and cwd differ * from the live replay run; the non-empty sentinel for missing cwd avoids accidental empty- * string replacement. * * @param fixture The committed `session.jsonl` content. * @returns The fixture's own volatile values, ready for {@link normalizeSessionLog}. */ export function fixtureContext(fixture: string): NormalizeContext { const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } return { sessionIds: typeof header.id === 'string' ? [header.id] : [], cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', } } /** * The `data.header` payload of every `request/header` event in a session * JSONL, in log order, with the log's volatile values scrubbed first * ({@link normalizeSessionLog}) so headers harvested from different runs — * each embedding its own generated cwd in the composed prompt — compare on equal * footing. * * @param rawLog The session `.jsonl` content to extract headers from. * @param ctx The volatile values of the run that produced it. * @returns The normalized `data.header` payloads, in log order. */ export function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) .filter(record => record.type === 'request/header') .map(record => record.data?.header) } /** * The normalized string-valued system prompts carried by request headers in a * session JSONL, in log order. Headers without a string prompt are omitted so * callers can assert one prompt per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized system prompts, in header order. */ export function normalizedSystemPrompts(rawLog: string, ctx: NormalizeContext): string[] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const system = (header as { system?: unknown }).system return typeof system === 'string' ? [system] : [] }) } /** * The normalized tool-schema arrays carried by request headers in a session * JSONL, in log order. Headers without an array-valued tools field are omitted * so callers can assert one schema set per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized initial tool-schema arrays, in header order. */ export function normalizedToolSchemas(rawLog: string, ctx: NormalizeContext): unknown[][] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const tools = (header as { tools?: unknown }).tools return Array.isArray(tools) ? [tools] : [] }) } /** The structured contents of a tool-schema sidecar. */ export interface ToolSchemasSnapshot { /** The complete tool schemas from the pinned request header. */ initial: unknown[] /** Complete tool schemas from subsequent changed-header snapshots. */ changes: unknown[][] } /** * Render the full tool-schema sequence as canonical, readable JSON. * * @param initial The pinned request header's complete tool schemas. * @param changes Complete tool schemas from later changed headers. * @returns A pretty-printed JSON snapshot ending in one newline. */ export function formatToolSchemasSnapshot(initial: readonly unknown[], changes: readonly unknown[][] = []): string { return `${JSON.stringify({ initial, changes }, null, 2)}\n` } /** * Parse and validate the stable top-level shape of a tool-schema sidecar. * * @param snapshot The JSON sidecar text. * @returns Its initial and changed-header schema sets. */ export function parseToolSchemasSnapshot(snapshot: string): ToolSchemasSnapshot { const parsed = JSON.parse(snapshot) as unknown if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed)) { throw new Error('acp-snapshot: tool-schema snapshot must be an object') } const { initial, changes } = parsed as { initial?: unknown; changes?: unknown } if (!Array.isArray(initial) || !Array.isArray(changes) || !changes.every(Array.isArray)) { throw new Error('acp-snapshot: tool-schema snapshot must carry array-valued initial and changes fields') } return { initial, changes } } /** * Restore one sidecar schema set into a tokenized pinned header. * * @param header The parsed request header carrying `tools: "{{tools}}"`. * @param schemas The complete schemas for this full header snapshot. * @returns A copy of the header with its complete schemas restored. */ export function restorePinnedToolSchemas(header: unknown, schemas: readonly unknown[]): unknown { if (header === null || typeof header !== 'object' || Array.isArray(header)) { throw new Error('acp-snapshot: pinned request header must be an object') } if ((header as { tools?: unknown }).tools !== TOOLS_TOKEN) { throw new Error(`acp-snapshot: pinned request header tools must equal ${TOOLS_TOKEN}`) } return { ...header, tools: schemas } } /** * Render a normalized prompt as a repository-friendly Markdown snapshot. * Prompt text is unchanged except that a missing terminal newline is added so * the committed file follows the repository newline contract. * * @param prompt The normalized system prompt. * @param changes Full normalized prompts from later changed-header snapshots. * @returns Markdown snapshot text ending in a newline. */ export function formatSystemPromptSnapshot( prompt: string, changes: readonly string[] = [], ): string { let snapshot = prompt.endsWith('\n') ? prompt : `${prompt}\n` for (const [index, change] of changes.entries()) { snapshot += `\n\n\n` snapshot += change.endsWith('\n') ? change : `${change}\n` } return snapshot } /** Return the initial-prompt portion of a possibly multi-header snapshot. */ function initialSystemPromptSnapshot(snapshot: string): string { const marker = snapshot.indexOf('\n