/** * Keyless-by-default ACP snapshot suite factory. Each scenario drives the real subprocess and * compares normalized stdout; comparable session fixtures are both replay input and expected * output. Record mode refreshes reproducible model scenarios from the live API, while refresh * mode replays committed scripts and rewrites derived artifacts without a key. * * Exactly one scenario per header-composition class pins the system prompt and tool schemas in * dedicated sidecars. Every live header is checked against that pin, so session-dependent * composition must declare a separate class instead of escaping coverage. * @module @deepseek-ai/dsh-acp-snapshot/suite */ import { readFile, readdir, writeFile } from 'node:fs/promises' import { existsSync } from 'node:fs' import { join } from 'node:path' import { describe, expect, it } from 'vitest' import { type AgentUnderTest, type HarvestedLog, type InputScript, runScenario } from './harness.ts' import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders, scrubSystemPrompts, scrubToolSchemas, } from './normalize.ts' /** The readable system-prompt snapshot beside each header-pinning fixture. */ const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.golden.md' /** The structured tool-schema snapshot beside each header-pinning fixture. */ const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.golden.json' /** Stable session-log token standing in for the sidecar's initial schemas. */ const TOOLS_TOKEN = '{{tools}}' /** A snapshot scenario and how its fixtures are produced. */ export interface Scenario { name: string /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ hasModelTurn: boolean /** * Whether the run persists a comparable session log to diff against the * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn * always produces a log worth comparing). Set it independently for a scenario * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` * events but never calls the model. */ comparesLog?: boolean /** * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` * from the LIVE API. `recorded` scenarios are model-driven and reproducible; * `authored` scenarios (fixtures hand-written or hand-harvested — e.g. a * provider error or a cancel the live API can't be coaxed into * deterministically, a deterministic hook scenario, or a scripted repetition * a live model won't reproduce) are NEVER re-recorded. */ recorded: boolean /** * Whether replay is driven by a hand-written `replay.override.json` sidecar * (a `ReplayEntry[]` that REPLACES the script derived from `session.jsonl`) * — the throw/hang cases chunks cannot express. The fixture guard requires * the sidecar exactly when this is set: the harness forwards the file purely * on existence, so an unregistered stray sidecar would silently replace the * derived script — the guard fails loud on either mismatch. Defaults to * false (replay derives from the fixture's `assistant/chunk` events). */ overridden?: boolean /** * How many SUBAGENT child sessions this scenario records beyond the top-level * one (0 for a single-session scenario). Each child rides in a sibling fixture * `session..jsonl` (1-based); replay forwards them to `dsh-llm-replay` so * each child session replays from its own script, and record mode writes the * harvested child logs back to those files. Defaults to 0. */ childSessions?: number /** * Whether this scenario is its header class's sole request-header pin. Dedicated sidecars own * the prompt and tool schemas, while every classmate is checked for equality. */ pinsHeader?: boolean /** * How many `request/header-delta` events this PINNING scenario's fixture * legitimately carries (default 0). A recorded mid-run header change — a * config-option switch rewriting a prompt section — is part of the pinned * surface, with readable prompt text in Markdown; any OTHER count * still fails, so fixture rot stays caught. Meaningless off the pin (the * live uniformity guard keeps non-pinning scenarios delta-free). */ expectedHeaderDeltas?: number /** * Which header-composition class this scenario belongs to. Scenarios that * boot the same config compose the same header; each class has exactly one * {@link pinsHeader} scenario, and the uniformity guard compares every * other member against ITS class's pin. Defaults to `'default'`; a * scenario booting an alternate config ({@link configPath}) whose tool * list or prompt sections differ by construction carries its own class. */ headerClass?: string /** * Alternate LIVE config path (absolute) this scenario boots instead of * {@link AgentUnderTest.configPath} — an overlay composing a different * tree (its basename must still end in `cordis.yml` so the bin's replay * swap finds the sibling `*cordis.snapshot.yml`). A scenario whose * overlay changes the composed header also needs its own * {@link headerClass}. */ configPath?: string } /** One suite's inputs: the agent to boot, where its fixtures live, and its scenario table. */ export interface SnapshotSuiteOptions { /** The agent composition every scenario boots. */ agent: AgentUnderTest /** Absolute path of the suite's `snapshots/` directory (one subdir per scenario). */ snapshotsDir: string /** The scenario table; exactly one entry per header class must set `pinsHeader`. */ scenarios: Scenario[] /** * `replay` (keyless, the default tier), `record` (live API; re-records the * `recorded` scenarios' fixtures and refreshes the Vitest goldens under * `--update`), or `refresh` (keyless replay that rewrites stdout goldens and * comparable session fixtures from the replay run). The caller derives this * from `$DSH_SNAPSHOT` — env reading stays outside this library. */ mode: 'replay' | 'record' | 'refresh' } /** * The sibling child-fixture paths for a scenario (`session.1.jsonl` …). * * @param dir The scenario's snapshots directory (`/`). * @param childSessions How many subagent child sessions the scenario records. * @returns One path per child, 1-based, in fixture order. */ export function childFixturePaths(dir: string, childSessions: number): string[] { return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) } /** * Derive normalization values from a fixture's own session header. Recorded ids and cwd differ * from the live replay run; the non-empty sentinel for missing cwd avoids accidental empty- * string replacement. * * @param fixture The committed `session.jsonl` content. * @returns The fixture's own volatile values, ready for {@link normalizeSessionLog}. */ export function fixtureContext(fixture: string): NormalizeContext { const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } return { sessionIds: typeof header.id === 'string' ? [header.id] : [], cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', } } /** * The `data.header` payload of every `request/header` event in a session * JSONL, in log order, with the log's volatile values scrubbed first * ({@link normalizeSessionLog}) so headers harvested from different runs — * each embedding its own temp cwd in the composed prompt — compare on equal * footing. * * @param rawLog The session `.jsonl` content to extract headers from. * @param ctx The volatile values of the run that produced it. * @returns The normalized `data.header` payloads, in log order. */ export function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) .filter(record => record.type === 'request/header') .map(record => record.data?.header) } /** * The normalized string-valued system prompts carried by request headers in a * session JSONL, in log order. Headers without a string prompt are omitted so * callers can assert one prompt per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized system prompts, in header order. */ export function normalizedSystemPrompts(rawLog: string, ctx: NormalizeContext): string[] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const system = (header as { system?: unknown }).system return typeof system === 'string' ? [system] : [] }) } /** * The normalized tool-schema arrays carried by request headers in a session * JSONL, in log order. Headers without an array-valued tools field are omitted * so callers can assert one schema set per header explicitly. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized initial tool-schema arrays, in header order. */ export function normalizedToolSchemas(rawLog: string, ctx: NormalizeContext): unknown[][] { return normalizedHeaders(rawLog, ctx).flatMap((header) => { if (header === null || typeof header !== 'object') return [] const tools = (header as { tools?: unknown }).tools return Array.isArray(tools) ? [tools] : [] }) } /** * Extract normalized tool-schema edits from request-header deltas in log order. * Deltas without an object-valued tools edit are omitted; their remaining * structure stays pinned in the session JSONL. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized tool-schema edits, in event order. */ export function normalizedToolSchemaDeltas(rawLog: string, ctx: NormalizeContext): unknown[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) .map(line => JSON.parse(line) as { type?: unknown; data?: { tools?: unknown } }) .filter(record => record.type === 'request/header-delta') .flatMap((record) => { const tools = record.data?.tools return tools !== null && typeof tools === 'object' && !Array.isArray(tools) ? [tools] : [] }) } /** The structured contents of a tool-schema sidecar. */ export interface ToolSchemasSnapshot { /** The complete tool schemas from the pinned request header. */ initial: unknown[] /** Complete tool-schema edits from subsequent request-header deltas. */ deltas: unknown[] } /** * Render tool schemas and later schema edits as canonical, readable JSON. * * @param initial The pinned request header's complete tool schemas. * @param deltas Complete tool-schema edits from request-header deltas. * @returns A pretty-printed JSON snapshot ending in one newline. */ export function formatToolSchemasSnapshot(initial: readonly unknown[], deltas: readonly unknown[] = []): string { return `${JSON.stringify({ initial, deltas }, null, 2)}\n` } /** * Parse and validate the stable top-level shape of a tool-schema sidecar. * * @param snapshot The JSON sidecar text. * @returns Its initial schemas and schema deltas. */ export function parseToolSchemasSnapshot(snapshot: string): ToolSchemasSnapshot { const parsed = JSON.parse(snapshot) as unknown if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed)) { throw new Error('acp-snapshot: tool-schema snapshot must be an object') } const { initial, deltas } = parsed as { initial?: unknown; deltas?: unknown } if (!Array.isArray(initial) || !Array.isArray(deltas)) { throw new Error('acp-snapshot: tool-schema snapshot must carry array-valued initial and deltas fields') } return { initial, deltas } } /** * Restore a sidecar's initial schemas into a tokenized pinned header. * * @param header The parsed request header carrying `tools: "{{tools}}"`. * @param snapshot The parsed tool-schema sidecar. * @returns A copy of the header with its complete initial schemas restored. */ export function restorePinnedToolSchemas(header: unknown, snapshot: ToolSchemasSnapshot): unknown { if (header === null || typeof header !== 'object' || Array.isArray(header)) { throw new Error('acp-snapshot: pinned request header must be an object') } if ((header as { tools?: unknown }).tools !== TOOLS_TOKEN) { throw new Error(`acp-snapshot: pinned request header tools must equal ${TOOLS_TOKEN}`) } return { ...header, tools: snapshot.initial } } /** One normalized system-prompt edit carried by a `request/header-delta`. */ export interface SystemPromptDeltaSnapshot { /** How many leading lines remain from the prior prompt. */ keepStart: number /** How many trailing lines remain from the prior prompt. */ keepEnd: number /** The normalized replacement lines inserted between the retained ranges. */ insert: string[] } /** * Extract normalized system-prompt edits from request-header deltas in log * order. Deltas without a well-formed system edit are omitted; their non-prompt * structure remains pinned in JSONL. * * @param rawLog The session `.jsonl` content to inspect. * @param ctx The volatile values of the run that produced it. * @returns The normalized system-prompt edits, in event order. */ export function normalizedSystemPromptDeltas(rawLog: string, ctx: NormalizeContext): SystemPromptDeltaSnapshot[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) .map(line => JSON.parse(line) as { type?: unknown; data?: { system?: unknown } }) .filter(record => record.type === 'request/header-delta') .flatMap((record) => { const system = record.data?.system if (system === null || typeof system !== 'object') return [] const { keepStart, keepEnd, insert } = system as { keepStart?: unknown; keepEnd?: unknown; insert?: unknown } if (typeof keepStart !== 'number' || typeof keepEnd !== 'number' || !Array.isArray(insert)) return [] if (!insert.every(line => typeof line === 'string')) return [] return [{ keepStart, keepEnd, insert: insert }] }) } /** * Render a normalized prompt as a repository-friendly Markdown snapshot. * Prompt text is unchanged except that a missing terminal newline is added so * the committed file follows the repository newline contract. * * @param prompt The normalized system prompt. * @param deltas Normalized prompt edits to append as readable sections. * @returns Markdown snapshot text ending in a newline. */ export function formatSystemPromptSnapshot( prompt: string, deltas: readonly SystemPromptDeltaSnapshot[] = [], ): string { let snapshot = prompt.endsWith('\n') ? prompt : `${prompt}\n` for (const [index, delta] of deltas.entries()) { snapshot += `\n\n\n` const insert = delta.insert.join('\n') snapshot += insert.endsWith('\n') ? insert : `${insert}\n` } return snapshot } /** Return the initial-prompt portion of a possibly delta-bearing snapshot. */ function initialSystemPromptSnapshot(snapshot: string): string { const marker = snapshot.indexOf('\n