Merge remote-tracking branch 'refs/remotes/origin/master' into worktree/retarget-pr961-20260808

# Conflicts:
#	.agents/notes/implemented/bug-fix/2026-07-27-stable-snapshot-refresh-volatiles.i18n.yaml
#	.agents/notes/implemented/bug-fix/2026-07-27-stable-snapshot-refresh-volatiles.zh.md
#	apps/web/tests/scaffold.ts
#	examples/tui-agent/tests/tui.snapshot.ts
#	packages/support/acp-snapshot/README.i18n.yaml
#	packages/support/acp-snapshot/README.md
#	packages/support/acp-snapshot/README.zh.md
This commit is contained in:
Tianyi Cui
2026-08-08 01:49:04 +08:00
4422 changed files with 272205 additions and 60192 deletions

View File

@@ -43,14 +43,24 @@ const WAIT_POLL_INTERVAL_MS = 10
* reference, since a committed file cannot know the id in advance.
*
* `promptAndCancel` starts a prompt without awaiting completion, waits for a
* readiness condition, then cancels and awaits completion. `waitForFile`
* observes a cwd-relative marker; the default observes the durable turn start.
* readiness condition, then cancels and awaits completion. Its optional
* `waitForFile` observes a cwd-relative marker; otherwise it waits for the
* durable turn start. The standalone `waitForFile` holds the next script step
* behind the same marker.
* `promptAndWaitForAgentMessage` arms an exact text-chunk waiter before sending
* the prompt, then keeps the application live until that later update arrives.
* `waitForTurnStart` waits for an open durable turn, optionally at or beyond a
* specified turn number. `waitForTurnEnd` holds the subprocess open until the
* selected session's latest complete raw-JSONL turn boundary is `turn/end`.
* `waitForGoalPhase` waits for the latest durable goal snapshot to reach one phase.
* `waitForInboxMessage` waits for inserted inbox text containing a scenario marker.
* `waitForSubagentTurnEnd` waits until one background child has persisted a
* closed model-work turn after its own descriptor; child progress has no ACP
* update to wait on.
* `waitForTitleAfterTurnEnd` additionally waits for a later durable title.
* `waitForEventAfterTurnEnd` waits until a complete record of the given event
* type follows the latest closed turn — for scenarios whose asserted state
* (e.g. a goal pause) is appended only after cancellation reaches idle.
* A standalone `cancel` may also wait for a cwd-relative readiness marker.
* All wait timeouts default to 10s.
*/
@@ -66,9 +76,14 @@ export type InputStep =
text: string
waitForFile?: { path: string; timeoutMs?: number }
}
| { op: 'waitForFile'; path: string; timeoutMs?: number }
| { op: 'waitForTurnStart'; minimumTurn?: number; timeoutMs?: number }
| { op: 'waitForTurnEnd'; timeoutMs?: number }
| { op: 'waitForSubagentTurnEnd'; child?: number; minimumTurn?: number; timeoutMs?: number }
| { op: 'waitForGoalPhase'; phase: 'active' | 'paused' | 'blocked' | 'complete'; timeoutMs?: number }
| { op: 'waitForInboxMessage'; text: string; timeoutMs?: number }
| { op: 'waitForTitleAfterTurnEnd'; timeoutMs?: number }
| { op: 'waitForEventAfterTurnEnd'; type: string; timeoutMs?: number }
| { op: 'cancel'; waitForFile?: { path: string; timeoutMs?: number } }
/** A scenario's `input.json`: an ordered list of input steps. */
@@ -290,7 +305,11 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise
(id) => { sessionId = id },
(id, timeoutMs, minimumTurn) => waitForPersistedTurnStart(sessionsRoot, id, timeoutMs, minimumTurn),
(id, timeoutMs) => waitForPersistedTurnEnd(sessionsRoot, id, timeoutMs),
(child, timeoutMs, minimumTurn) => waitForPersistedChildTurnEnd(sessionsRoot, child, timeoutMs, minimumTurn),
(id, phase, timeoutMs) => waitForPersistedGoalPhase(sessionsRoot, id, phase, timeoutMs),
(id, text, timeoutMs) => waitForPersistedInboxMessage(sessionsRoot, id, text, timeoutMs),
(id, timeoutMs) => waitForPersistedTitleAfterTurnEnd(sessionsRoot, id, timeoutMs),
(id, type, timeoutMs) => waitForPersistedEventAfterTurnEnd(sessionsRoot, id, type, timeoutMs),
)
// A permission exchange happens while a step's request is in flight, so
// by the time the step settles any script bug it exposed is captured —
@@ -364,7 +383,11 @@ async function runStep(
setSessionId: (id: string) => void,
waitForTurnStart: (sessionId: string, timeoutMs?: number, minimumTurn?: number) => Promise<void>,
waitForTurnEnd: (sessionId: string, timeoutMs?: number) => Promise<void>,
waitForChildTurnEnd: (child: number, timeoutMs?: number, minimumTurn?: number) => Promise<void>,
waitForGoalPhase: (sessionId: string, phase: string, timeoutMs?: number) => Promise<void>,
waitForInboxMessage: (sessionId: string, text: string, timeoutMs?: number) => Promise<void>,
waitForTitleAfterTurnEnd: (sessionId: string, timeoutMs?: number) => Promise<void>,
waitForEventAfterTurnEnd: (sessionId: string, type: string, timeoutMs?: number) => Promise<void>,
): Promise<void> {
switch (step.op) {
case 'initialize':
@@ -436,18 +459,42 @@ async function runStep(
await promptDone
return
}
case 'waitForFile':
await waitForWorkspaceFile(cwd, step.path, step.timeoutMs)
return
case 'waitForTurnEnd': {
const sessionId = getSessionId()
if (sessionId === undefined) throw new Error('snapshot-harness: waitForTurnEnd before newSession')
await waitForTurnEnd(sessionId, step.timeoutMs)
return
}
case 'waitForSubagentTurnEnd':
await waitForChildTurnEnd(step.child ?? 1, step.timeoutMs, step.minimumTurn)
return
case 'waitForGoalPhase': {
const sessionId = getSessionId()
if (sessionId === undefined) throw new Error('snapshot-harness: waitForGoalPhase before newSession')
await waitForGoalPhase(sessionId, step.phase, step.timeoutMs)
return
}
case 'waitForInboxMessage': {
const sessionId = getSessionId()
if (sessionId === undefined) throw new Error('snapshot-harness: waitForInboxMessage before newSession')
await waitForInboxMessage(sessionId, step.text, step.timeoutMs)
return
}
case 'waitForTitleAfterTurnEnd': {
const sessionId = getSessionId()
if (sessionId === undefined) throw new Error('snapshot-harness: waitForTitleAfterTurnEnd before newSession')
await waitForTitleAfterTurnEnd(sessionId, step.timeoutMs)
return
}
case 'waitForEventAfterTurnEnd': {
const sessionId = getSessionId()
if (sessionId === undefined) throw new Error('snapshot-harness: waitForEventAfterTurnEnd before newSession')
await waitForEventAfterTurnEnd(sessionId, step.type, step.timeoutMs)
return
}
case 'waitForTurnStart': {
const sessionId = getSessionId()
if (sessionId === undefined) throw new Error('snapshot-harness: waitForTurnStart before newSession')
@@ -515,6 +562,95 @@ async function waitForPersistedTurnEnd(
}, { interval: WAIT_POLL_INTERVAL_MS, timeout: timeoutMs })
}
/**
* Wait until the Nth harvested child Session closes a model work turn.
*
* Harvest order matches `session.1.jsonl`, `session.2.jsonl`, and so on. A
* continuable child appends its descriptor after any inherited history and
* before accepting its first prompt, so only a later request header proves its
* own model work reached a closed turn.
*/
async function waitForPersistedChildTurnEnd(
root: string,
child: number,
timeoutMs = DEFAULT_WAIT_TIMEOUT_MS,
minimumTurn = 1,
): Promise<void> {
await vi.waitFor(async () => {
const log = (await harvestSessionLogs(root))[child]
if (log === undefined || !latestTurnIsClosed(log.content)
|| !hasRequestHeaderAfterDescriptor(log.content)
|| !hasClosedTurn(log.content, minimumTurn)) {
throw new Error(
`snapshot-harness: subagent child #${child} did not persist closed turn ${minimumTurn} within ${timeoutMs}ms`,
)
}
}, { interval: WAIT_POLL_INTERVAL_MS, timeout: timeoutMs })
}
/** Whether a raw session log contains the requested closed turn. */
function hasClosedTurn(content: string, turn: number): boolean {
return content.split('\n').filter(Boolean).some((line) => {
const event = JSON.parse(line) as { type?: unknown; data?: { turn?: unknown } }
return event.type === 'turn/end' && event.data?.turn === turn
})
}
/** Wait until the latest durable goal snapshot reaches one phase. */
async function waitForPersistedGoalPhase(
root: string,
sessionId: string,
phase: string,
timeoutMs = DEFAULT_WAIT_TIMEOUT_MS,
): Promise<void> {
await vi.waitFor(async () => {
const content = (await harvestSessionLogs(root)).find(log => log.id === sessionId)?.content
const matched = content?.split('\n').filter(Boolean).some((line) => {
const event = JSON.parse(line) as { type?: unknown; data?: { goal?: { phase?: unknown } } }
return event.type === 'goal/change' && event.data?.goal?.phase === phase
}) ?? false
if (!matched) {
throw new Error(`snapshot-harness: session "${sessionId}" did not persist goal phase "${phase}" within ${timeoutMs}ms`)
}
}, { interval: WAIT_POLL_INTERVAL_MS, timeout: timeoutMs })
}
/** Wait until an inserted inbox message contains scenario-owned text. */
async function waitForPersistedInboxMessage(
root: string,
sessionId: string,
text: string,
timeoutMs = DEFAULT_WAIT_TIMEOUT_MS,
): Promise<void> {
await vi.waitFor(async () => {
const log = (await harvestSessionLogs(root)).find(candidate => candidate.id === sessionId)
const matched = log?.content.split('\n').some((line) => {
if (line.length === 0) return false
const record = JSON.parse(line) as {
type?: unknown
data?: { inserted?: Array<{ content?: Array<{ type?: unknown; text?: unknown }> }> }
}
return record.type === 'agent/inbox/spliced' && record.data?.inserted?.some(message =>
message.content?.some(block => block.type === 'text'
&& typeof block.text === 'string' && block.text.includes(text))) === true
}) ?? false
if (!matched) {
throw new Error(`snapshot-harness: session "${sessionId}" did not persist expected inbox message within ${timeoutMs}ms`)
}
}, { interval: WAIT_POLL_INTERVAL_MS, timeout: timeoutMs })
}
/** Whether a child log contains model work after its own descriptor event. */
function hasRequestHeaderAfterDescriptor(content: string): boolean {
const events = content.slice(0, content.lastIndexOf('\n') + 1)
.split('\n')
.filter(line => line.length > 0)
.map(line => JSON.parse(line) as { type?: unknown })
const descriptor = events.findLastIndex(event => event.type === 'subagent/descriptor')
return descriptor >= 0
&& events.slice(descriptor + 1).some(event => event.type === 'request/header')
}
/** Wait until a complete provider or fallback title record follows the latest closed turn. */
async function waitForPersistedTitleAfterTurnEnd(
root: string,
@@ -529,6 +665,21 @@ async function waitForPersistedTitleAfterTurnEnd(
}, { interval: WAIT_POLL_INTERVAL_MS, timeout: timeoutMs })
}
/** Wait until a complete record of `type` follows the latest closed turn. */
async function waitForPersistedEventAfterTurnEnd(
root: string,
sessionId: string,
type: string,
timeoutMs = DEFAULT_WAIT_TIMEOUT_MS,
): Promise<void> {
await vi.waitFor(async () => {
const log = (await harvestSessionLogs(root)).find(candidate => candidate.id === sessionId)
if (log === undefined || !latestEventFollowsTurnEnd(log.content, type)) {
throw new Error(`snapshot-harness: session "${sessionId}" did not persist ${type} after turn/end within ${timeoutMs}ms`)
}
}, { interval: WAIT_POLL_INTERVAL_MS, timeout: timeoutMs })
}
/** Wait for a cwd-relative marker proving an external action reached readiness. */
async function waitForWorkspaceFile(
cwd: string,
@@ -557,6 +708,13 @@ function latestTitleFollowsTurnEnd(content: string): boolean {
return turnEnd >= 0 && complete.lastIndexOf('\n{"type":"session/title",') > turnEnd
}
/** Return whether a complete record of `type` occurs after the last complete turn end. */
function latestEventFollowsTurnEnd(content: string, type: string): boolean {
const complete = content.slice(0, content.lastIndexOf('\n') + 1)
const turnEnd = complete.lastIndexOf('\n{"type":"turn/end",')
return turnEnd >= 0 && complete.lastIndexOf(`\n{"type":"${type}",`) > turnEnd
}
/** Return the latest open turn number, validating the persisted boundary record. */
function latestOpenTurn(content: string): number | undefined {
const complete = content.slice(0, content.lastIndexOf('\n') + 1)

View File

@@ -197,7 +197,7 @@ function tokenizeFixtureString(value: string, ctx: NormalizeContext, basename: s
+ String.raw`(?=$|[\\/\s<>'"()\[\]{},;:!?=])`,
'g',
)
return exact.replace(absoluteCwd, CWD)
return exact.replace(absoluteCwd, CWD).split(`/private${CWD}`).join(CWD)
}
/** Recursively replace generated-cwd spellings while preserving every other JSON value. */

View File

@@ -40,6 +40,11 @@ const SYSTEM_PROMPT_SNAPSHOT = 'system-prompt.expected.md'
/** The structured tool-schema snapshot beside its owning header pin. */
const TOOL_SCHEMAS_SNAPSHOT = 'tool-schemas.expected.json'
/** Return the dedicated tool-schema sidecar for one child fixture index. */
function childToolSchemasSnapshot(index: number): string {
return `tool-schemas.${index}.expected.json`
}
/** The optional full Windows-native stdout transcript. */
const WINDOWS_STDOUT_SNAPSHOT = 'stdout.expected.windows.jsonl'
@@ -103,6 +108,13 @@ export interface Scenario {
* declare the same {@link expectedHeaderChanges}; meaningless off a pin.
*/
toolSchemasSource?: string
/**
* Child fixture indices whose own schema sequence is pinned separately,
* where `1` names `session.1.jsonl` and
* `tool-schemas.1.expected.json`. The class pin still owns every other
* request-header field.
*/
pinsChildToolSchemas?: readonly number[]
/**
* How many changed `request/header` snapshots this PINNING scenario's primary
* fixture legitimately carries (default 0). Their full prompt text is kept in
@@ -152,25 +164,37 @@ export interface Scenario {
* test is skipped on Windows; its fixtures stay guarded on every platform.
*/
posixOnly?: boolean
/**
* Whether the scenario boots a composition that needs a usable `pwsh`
* (the pwsh-tool-turn scenario). The run test is skipped when the suite's
* {@link SnapshotSuiteOptions.hasPwsh} probe is false; fixtures stay guarded
* on every platform.
*/
pwshOnly?: boolean
}
/**
* Whether a scenario's run test is skipped for this mode and host: record mode
* skips authored (non-`recorded`) scenarios, and {@link Scenario.posixOnly}
* scenarios skip on Windows.
* skips authored (non-`recorded`) scenarios, {@link Scenario.posixOnly}
* scenarios skip on Windows, and {@link Scenario.pwshOnly} scenarios skip
* when the caller's `hasPwsh` probe is false.
*
* @param scenario The scenario whose run test is being registered.
* @param recording Whether the suite runs in record mode.
* @param platform The running Node platform, injectable for unit coverage.
* @param hasPwsh The caller's pwsh-availability probe; `pwshOnly` scenarios
* skip unless it is true.
* @returns True when the scenario's run test must not execute.
*/
export function scenarioSkipped(
scenario: Scenario,
recording: boolean,
platform: NodeJS.Platform = process.platform,
hasPwsh?: boolean,
): boolean {
if (recording && !scenario.recorded) return true
return scenario.posixOnly === true && platform === 'win32'
if (scenario.posixOnly === true && platform === 'win32') return true
return scenario.pwshOnly === true && hasPwsh !== true
}
/** One stdout expected output selected for a platform run. */
@@ -211,6 +235,11 @@ export interface SnapshotSuiteOptions {
* from `$DSH_SNAPSHOT` — env reading stays outside this library.
*/
mode: 'replay' | 'record' | 'refresh'
/**
* Whether a real `pwsh` executable is available on this host (the probe the
* caller owns; `pwshOnly` scenarios skip when this is not true).
*/
hasPwsh?: boolean
}
/** One scenario's generated claim on a shared snapshot file. */
@@ -1052,8 +1081,9 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
scenarioSuite('snapshot scenarios', () => {
for (const scenario of scenarios) {
// In RECORD mode, only re-run the `recorded` (live-API) scenarios; the `authored` ones
// (sidecar-driven errors/cancel) are never re-recorded. `posixOnly` scenarios skip on Windows.
it.skipIf(scenarioSkipped(scenario, RECORDING))(`snapshot: ${scenario.name} matches the expected outputs`, async ({ expect }) => {
// (sidecar-driven errors/cancel) are never re-recorded. `posixOnly` scenarios skip on Windows;
// `pwshOnly` scenarios skip when the caller's `hasPwsh` probe is false.
it.skipIf(scenarioSkipped(scenario, RECORDING, process.platform, options.hasPwsh))(`snapshot: ${scenario.name} matches the expected outputs`, async ({ expect }) => {
const dir = join(snapshotsDir, scenario.name)
const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript
const overrideFile = join(dir, 'replay.override.json')
@@ -1099,6 +1129,8 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
cwdAliases: result.cwdAliases,
}
const childSchemaPins = new Set(scenario.pinsChildToolSchemas ?? [])
// Record writes live model fixtures; keyless refresh writes every comparable replayed
// fixture. Pinning JSONL keeps prefixes but moves prompts and schemas into sidecars.
const scrub = scenario.pinsHeader === true
@@ -1177,6 +1209,18 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
claimSharedSnapshot(schemaClaims, schemaPath, scenario.name, toolSchemasSnapshot)
await writeFile(schemaPath, toolSchemasSnapshot)
}
for (const index of childSchemaPins) {
const log = result.sessionLogs[index]
expect(log, `${mode}: no child session log at index ${index} to snapshot schemas from`)
.toBeDefined()
const schemaSets = normalizedToolSchemas((log as HarvestedLog).content, ctx)
expect(schemaSets.length, `${mode}: child ${index} produced no tool schemas to snapshot`)
.toBeGreaterThan(0)
await writeFile(join(dir, childToolSchemasSnapshot(index)), formatToolSchemasSnapshot(
schemaSets[0] as unknown[],
schemaSets.slice(1),
))
}
}
for (const expected of stdoutExpectedVariants(scenario)) {
@@ -1230,7 +1274,14 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
header,
pinnedSchemaSets[index] as unknown[],
))
const childPinnedSchemas = new Map<number, unknown[][]>()
for (const index of childSchemaPins) {
const sidecar = await readFile(join(dir, childToolSchemasSnapshot(index)), 'utf8')
const parsed = parseToolSchemasSnapshot(sidecar)
childPinnedSchemas.set(index, [parsed.initial, ...parsed.changes])
}
for (const [logIndex, log] of result.sessionLogs.entries()) {
const childSchemas = childPinnedSchemas.get(logIndex)
const expectedChanges = scenario.pinsHeader === true && logIndex === 0
? scenario.expectedHeaderChanges ?? 0
: 0
@@ -1243,8 +1294,15 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
.toBe(headers.length)
expect(schemaSets.length, `session ${log.id}: every request/header must carry an array-valued tools field`)
.toBe(headers.length)
if (childSchemas !== undefined) {
expect(childSchemas.length, `session ${log.id}: ${childToolSchemasSnapshot(logIndex)} has an unexpected tool-schema count`)
.toBe(schemaSets.length)
}
for (const [k, header] of headers.entries()) {
const expected = expectedChanges > 0 ? pinnedHeaders[k] : pinnedHeaders[0]
const classPin = expectedChanges > 0 ? pinnedHeaders[k] : pinnedHeaders[0]
const expected = childSchemas === undefined
? classPin
: { ...classPin as Record<string, unknown>, tools: childSchemas[k] }
expect(header, `session ${log.id}: request/header #${k + 1} diverged from the pinned (${pinningScenario.name}) header`)
.toEqual(expected)
if (expectedChanges === 0) {
@@ -1282,8 +1340,16 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
it('every registered scenario has its required fixture files', async () => {
// Every scenario needs input, stdout, a primary session fixture, and matching optional sidecars.
for (const { name, overridden, pinsNativeWindowsStdout } of scenarios) {
for (const { name, overridden, pinsNativeWindowsStdout, pinsChildToolSchemas } of scenarios) {
const dir = join(snapshotsDir, name)
const declaredChildPins = new Set(pinsChildToolSchemas ?? [])
const childSidecars = (await readdir(dir, { withFileTypes: true }))
.filter(entry => entry.isFile())
.map(entry => /^tool-schemas\.([1-9]\d*)\.expected\.json$/.exec(entry.name))
.filter((match): match is RegExpExecArray => match !== null)
.map(match => Number(match[1]))
expect(new Set(childSidecars), `${name}: child tool-schema sidecars must match \`pinsChildToolSchemas\``)
.toEqual(declaredChildPins)
expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true)
expect(existsSync(join(dir, 'stdout.expected.jsonl')), `${name}/stdout.expected.jsonl`).toBe(true)
expect(
@@ -1366,9 +1432,30 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
assertUniqueSnapshotContents('tool-schema', schemas)
})
it('every committed JSONL has valid tool results and canonical header storage', async () => {
it('every declared child tool-schema sidecar is canonical and names a real child', async () => {
for (const scenario of scenarios) {
const pins = scenario.pinsChildToolSchemas ?? []
if (pins.length === 0) continue
const dir = join(snapshotsDir, scenario.name)
const files = await sessionFixtures(dir)
for (const index of pins) {
expect(files[index], `${scenario.name}: child schema pin ${index} must name an existing session.<n>.jsonl fixture`)
.toBeDefined()
const file = childToolSchemasSnapshot(index)
const sidecar = await readFile(join(dir, file), 'utf8')
const parsed = parseToolSchemasSnapshot(sidecar)
expect(sidecar, `${scenario.name}/${file} must use canonical JSON formatting`)
.toBe(formatToolSchemasSnapshot(parsed.initial, parsed.changes))
expect(parsed.initial.length, `${scenario.name}/${file} must pin at least one schema`)
.toBeGreaterThan(0)
}
}
})
it('every committed JSONL has valid tool results and canonical fixture storage', async () => {
// Prompts and schemas always leave JSONL. Header pins retain prefixes;
// every other fixture tokenizes those too. Fixed-point checks make both
// every other fixture tokenizes those too. Portable cwd tokens never
// retain a platform realpath prefix. Fixed-point checks make these
// storage rules fail loud.
for (const scenario of scenarios) {
const dir = join(snapshotsDir, scenario.name)
@@ -1377,6 +1464,8 @@ export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void {
const fixture = await readFile(join(dir, file), 'utf8')
expect(unknownToolCallIds(fixture), `${scenario.name}/${file} contains UNKNOWN_TOOL`)
.toEqual([])
expect(fixture, `${scenario.name}/${file} carries a non-canonical macOS cwd token`)
.not.toContain('/private{{cwd}}')
expect(scrubSystemPrompts(fixture), `${scenario.name}/${file} carries an unscrubbed system prompt`)
.toEqual(fixture)
expect(scrubToolSchemas(fixture), `${scenario.name}/${file} carries unscrubbed tool schemas`)