/** * Keyless-by-default ACP snapshot suite factory. Each scenario drives the real * subprocess and compares normalized stdout; comparable session fixtures are * both replay input and expected output. Record mode refreshes reproducible * model scenarios from the live API, while refresh mode replays committed * scripts and rewrites derived artifacts without a key. * Replay scenarios run concurrently because each subprocess owns unique temp * cwd and persistence roots and reads only committed fixtures. Record and * refresh stay serial while writing. * * Exactly one scenario per header-composition class pins the full prompt and * tool-schema sequences in dedicated sidecars. Every live header is checked * against that pin, so session-dependent composition must declare a separate * class instead of escaping coverage. * @module @deepseek-ai/dsh-acp-snapshot/suite */ import { readFile, readdir, writeFile } from 'node:fs/promises' import { existsSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { type AgentUnderTest, type HarvestedLog, type InputScript, runScenario } from './harness.ts' import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders, scrubSystemPrompts, scrubToolSchemas, } from './normalize.ts' /** The readable system-prompt snapshot beside each header-pinning fixture. */ const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.golden.md' /** The structured tool-schema snapshot beside each header-pinning fixture. */ const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.golden.json' /** Stable session-log token standing in for the sidecar's initial schemas. */ const TOOLS_TOKEN = '{{tools}}' /** A snapshot scenario and how its fixtures are produced. */ export interface Scenario { name: string /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ hasModelTurn: boolean /** * Whether the run persists a comparable session log to diff against the * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn * always produces a log worth comparing). Set it independently for a scenario * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` * events but never calls the model. */ comparesLog?: boolean /** * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` * from the LIVE API. `recorded` scenarios are model-driven and reproducible; * `authored` scenarios (fixtures hand-written or hand-harvested — e.g. a * provider error or a cancel the live API can't be coaxed into * deterministically, a deterministic hook scenario, or a scripted repetition * a live model won't reproduce) are NEVER re-recorded. */ recorded: boolean /** * Whether replay is driven by a hand-written `replay.override.json` sidecar * (a `ReplayEntry[]` that REPLACES the script derived from `session.jsonl`) * — the throw/hang cases chunks cannot express. The fixture guard requires * the sidecar exactly when this is set: the harness forwards the file purely * on existence, so an unregistered stray sidecar would silently replace the * derived script — the guard fails loud on either mismatch. Defaults to * false (replay derives from the fixture's `assistant/chunk` events). */ overridden?: boolean /** * How many SUBAGENT child sessions this scenario records beyond the top-level * one (0 for a single-session scenario). Each child rides in a sibling fixture * `session..jsonl` (1-based); replay forwards them to `dsh-llm-replay` so * each child session replays from its own script, and record mode writes the * harvested child logs back to those files. Defaults to 0. */ childSessions?: number /** * Whether this scenario is its header class's sole request-header pin. Dedicated sidecars own * the prompt and tool schemas, while every classmate is checked for equality. */ pinsHeader?: boolean /** * How many changed `request/header` snapshots this PINNING scenario's primary * fixture legitimately carries (default 0). Their full prompt text is kept in * the readable Markdown pin; any other count fails. Meaningless off the pin. */ expectedHeaderChanges?: number /** * Which header-composition class this scenario belongs to. Scenarios that * boot the same config compose the same header; each class has exactly one * {@link pinsHeader} scenario, and the uniformity guard compares every * other member against ITS class's pin. Defaults to `'default'`; a * scenario booting an alternate config ({@link configPath}) whose tool * list or prompt sections differ by construction carries its own class. */ headerClass?: string /** * Alternate LIVE config path (absolute) this scenario boots instead of * {@link AgentUnderTest.configPath} — an overlay composing a different * tree (its basename must still end in `cordis.yml` so the bin's replay * swap finds the sibling `*cordis.snapshot.yml`). A scenario whose * overlay changes the composed header also needs its own * {@link headerClass}. */ configPath?: string } /** One suite's inputs: the agent to boot, where its fixtures live, and its scenario table. */ export interface SnapshotSuiteOptions { /** The agent composition every scenario boots. */ agent: AgentUnderTest /** Absolute path of the suite's `snapshots/` directory (one subdir per scenario). */ snapshotsDir: string /** The scenario table; exactly one entry per header class must set `pinsHeader`. */ scenarios: Scenario[] /** * `replay` (keyless, the default tier), `record` (live API; re-records the * `recorded` scenarios' fixtures and refreshes the Vitest goldens under * `--update`), or `refresh` (keyless replay that rewrites stdout goldens and * comparable session fixtures from the replay run). The caller derives this * from `$DSH_SNAPSHOT` — env reading stays outside this library. */ mode: 'replay' | 'record' | 'refresh' } /** * The sibling child-fixture paths for a scenario (`session.1.jsonl` …). * * @param dir The scenario's snapshots directory (`/`). * @param childSessions How many subagent child sessions the scenario records. * @returns One path per child, 1-based, in fixture order. */ export function childFixturePaths(dir: string, childSessions: number): string[] { return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) } /** * Derive normalization values from a fixture's own session header. Recorded ids and cwd differ * from the live replay run; the non-empty sentinel for missing cwd avoids accidental empty- * string replacement. * * @param fixture The committed `session.jsonl` content. * @returns The fixture's own volatile values, ready for {@link normalizeSessionLog}. */ export function fixtureContext(fixture: string): NormalizeContext { const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } return { sessionIds: typeof header.id === 'string' ? [header.id] : [], cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', } } /** * The `data.header` payload of every `request/header` event in a session * JSONL, in log order, with the log's volatile values scrubbed first * ({@link normalizeSessionLog}) so headers harvested from different runs — * each embedding its own temp cwd in the composed prompt — compare on equal * footing. * * @param rawLog The session `.jsonl` content to extract headers from. * @param ctx The volatile values of the run that produced it. * @returns The normalized `data.header` payloads, in log order. */ export function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) .filter(record => record.type === 'request/header') .map(record => record.data?.header) } /** * The normalized string-valued system prompts carried by request headers in a * session JSONL, in log order. Headers without a string prompt are omitted so * callers can assert one prompt per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized system prompts, in header order. */ export function normalizedSystemPrompts(rawLog: string, ctx: NormalizeContext): string[] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const system = (header as { system?: unknown }).system return typeof system === 'string' ? [system] : [] }) } /** * The normalized tool-schema arrays carried by request headers in a session * JSONL, in log order. Headers without an array-valued tools field are omitted * so callers can assert one schema set per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized initial tool-schema arrays, in header order. */ export function normalizedToolSchemas(rawLog: string, ctx: NormalizeContext): unknown[][] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const tools = (header as { tools?: unknown }).tools return Array.isArray(tools) ? [tools] : [] }) } /** The structured contents of a tool-schema sidecar. */ export interface ToolSchemasSnapshot { /** The complete tool schemas from the pinned request header. */ initial: unknown[] /** Complete tool schemas from subsequent changed-header snapshots. */ changes: unknown[][] } /** * Render the full tool-schema sequence as canonical, readable JSON. * * @param initial The pinned request header's complete tool schemas. * @param changes Complete tool schemas from later changed headers. * @returns A pretty-printed JSON snapshot ending in one newline. */ export function formatToolSchemasSnapshot(initial: readonly unknown[], changes: readonly unknown[][] = []): string { return `${JSON.stringify({ initial, changes }, null, 2)}\n` } /** * Parse and validate the stable top-level shape of a tool-schema sidecar. * * @param snapshot The JSON sidecar text. * @returns Its initial and changed-header schema sets. */ export function parseToolSchemasSnapshot(snapshot: string): ToolSchemasSnapshot { const parsed = JSON.parse(snapshot) as unknown if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed)) { throw new Error('acp-snapshot: tool-schema snapshot must be an object') } const { initial, changes } = parsed as { initial?: unknown; changes?: unknown } if (!Array.isArray(initial) || !Array.isArray(changes) || !changes.every(Array.isArray)) { throw new Error('acp-snapshot: tool-schema snapshot must carry array-valued initial and changes fields') } return { initial, changes } } /** * Restore one sidecar schema set into a tokenized pinned header. * * @param header The parsed request header carrying `tools: "{{tools}}"`. * @param schemas The complete schemas for this full header snapshot. * @returns A copy of the header with its complete schemas restored. */ export function restorePinnedToolSchemas(header: unknown, schemas: readonly unknown[]): unknown { if (header === null || typeof header !== 'object' || Array.isArray(header)) { throw new Error('acp-snapshot: pinned request header must be an object') } if ((header as { tools?: unknown }).tools !== TOOLS_TOKEN) { throw new Error(`acp-snapshot: pinned request header tools must equal ${TOOLS_TOKEN}`) } return { ...header, tools: schemas } } /** * Render a normalized prompt as a repository-friendly Markdown snapshot. * Prompt text is unchanged except that a missing terminal newline is added so * the committed file follows the repository newline contract. * * @param prompt The normalized system prompt. * @param changes Full normalized prompts from later changed-header snapshots. * @returns Markdown snapshot text ending in a newline. */ export function formatSystemPromptSnapshot( prompt: string, changes: readonly string[] = [], ): string { let snapshot = prompt.endsWith('\n') ? prompt : `${prompt}\n` for (const [index, change] of changes.entries()) { snapshot += `\n\n\n` snapshot += change.endsWith('\n') ? change : `${change}\n` } return snapshot } /** Return the initial-prompt portion of a possibly multi-header snapshot. */ function initialSystemPromptSnapshot(snapshot: string): string { const marker = snapshot.indexOf('\n