Merge remote-tracking branch 'origin/master' into worktree/pr504-retarget-latest-master
# Conflicts: # docs/event-producer-consumer.md # examples/tui-agent/README.md # examples/tui-agent/composition.md # examples/tui-agent/cordis.yml # examples/tui-agent/tests/snapshots/multi-turn-conversation/terminal.expected.txt # examples/tui-agent/tests/tui-keyless-smoke.e2e.ts # packages/ui/tui/README.md # vitest.config.ts
This commit is contained in:
@@ -24,6 +24,8 @@ import { basename, dirname, join, delimiter } from 'node:path'
|
||||
import {
|
||||
ClientSideConnection,
|
||||
PROTOCOL_VERSION,
|
||||
type CreateElicitationRequest,
|
||||
type CreateElicitationResponse,
|
||||
type RequestPermissionRequest,
|
||||
type RequestPermissionResponse,
|
||||
type SessionNotification,
|
||||
@@ -59,6 +61,8 @@ export type InputStep =
|
||||
waitForToolCallUpdate?: string
|
||||
}
|
||||
| { op: 'cancel' }
|
||||
| { op: 'setMode'; modeId: string }
|
||||
| { op: 'setModeExpectError'; modeId: string }
|
||||
| { op: 'setConfigOption'; configId: string; value: string }
|
||||
| { op: 'setConfigOptionExpectError'; configId: string; value: string }
|
||||
|
||||
@@ -78,6 +82,16 @@ export interface InputScript {
|
||||
* agent itself just sees `cancelled`, so it cannot absorb the bug).
|
||||
*/
|
||||
permissionAnswers?: PermissionAnswer[]
|
||||
/**
|
||||
* Ordered answers for the agent's `elicitation/create` round-trips (the
|
||||
* ask_user_question / plan-review forms), consumed FIFO — the Nth request
|
||||
* gets the Nth answer. Exhaustion (or no queue) answers `cancel`, the same
|
||||
* fail-closed stub an elicitation-free scenario relies on. Unlike permission
|
||||
* kinds, the scripted strings are not validated against the offered form —
|
||||
* a stray `choice` reaches the agent verbatim, which reads it as a custom
|
||||
* (non-consenting) answer, so a scenario bug fails safe in the transcript.
|
||||
*/
|
||||
elicitationAnswers?: ElicitationAnswer[]
|
||||
}
|
||||
|
||||
/** One scripted answer to a permission request: which offered option kind to select. */
|
||||
@@ -86,6 +100,16 @@ export interface PermissionAnswer {
|
||||
kind: 'allow_once' | 'allow_always' | 'reject_once' | 'reject_always'
|
||||
}
|
||||
|
||||
/** One scripted answer to an elicitation form (accept with choice/custom content, or cancel). */
|
||||
export interface ElicitationAnswer {
|
||||
/** Accept the form with the content below, or cancel it. */
|
||||
action: 'accept' | 'cancel'
|
||||
/** The selected option label (the form's `choice` field). */
|
||||
choice?: string
|
||||
/** Free-form text (the form's `custom` field). */
|
||||
custom?: string
|
||||
}
|
||||
|
||||
/** One harvested session log plus the identifying facts off its header line. */
|
||||
export interface HarvestedLog {
|
||||
/** The recorded session id (header `id`). */
|
||||
@@ -215,6 +239,8 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise
|
||||
// Permission answers are consumed FIFO across the whole run; exhaustion
|
||||
// falls back to `cancelled` so approval-free scenarios keep the plain stub.
|
||||
const permissionQueue = [...input.permissionAnswers ?? []]
|
||||
// Elicitation answers mirror the permission queue: FIFO, cancel on exhaustion.
|
||||
const elicitationQueue = [...input.elicitationAnswers ?? []]
|
||||
// A scenario bug detected inside a client callback (a scripted permission
|
||||
// kind the agent never offered). It cannot fail the run from in there: a
|
||||
// callback throw only becomes a JSON-RPC error RESPONSE to the agent, and
|
||||
@@ -244,6 +270,17 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise
|
||||
}
|
||||
return Promise.resolve({ outcome: { outcome: 'selected', optionId: option.optionId } })
|
||||
},
|
||||
createElicitation(_params: CreateElicitationRequest): Promise<CreateElicitationResponse> {
|
||||
const answer = elicitationQueue.shift()
|
||||
if (answer === undefined || answer.action !== 'accept') return Promise.resolve({ action: 'cancel' })
|
||||
return Promise.resolve({
|
||||
action: 'accept',
|
||||
content: {
|
||||
...answer.choice !== undefined ? { choice: answer.choice } : {},
|
||||
...answer.custom !== undefined ? { custom: answer.custom } : {},
|
||||
},
|
||||
})
|
||||
},
|
||||
})
|
||||
const active = launched
|
||||
await active.spawned
|
||||
@@ -399,6 +436,24 @@ async function runStep(
|
||||
await client.cancel({ sessionId })
|
||||
return
|
||||
}
|
||||
case 'setMode': {
|
||||
const sessionId = getSessionId()
|
||||
if (sessionId === undefined) throw new Error('snapshot-harness: setMode before newSession')
|
||||
await client.setSessionMode({ sessionId, modeId: step.modeId })
|
||||
return
|
||||
}
|
||||
case 'setModeExpectError': {
|
||||
const sessionId = getSessionId()
|
||||
if (sessionId === undefined) throw new Error('snapshot-harness: setModeExpectError before newSession')
|
||||
// The bridge rejects an unknown/uncomposed mode id with invalidParams;
|
||||
// that rejection IS the expected wire behavior — swallow it so the run
|
||||
// completes and the error frame is captured in the transcript.
|
||||
await client.setSessionMode({ sessionId, modeId: step.modeId }).then(
|
||||
() => { throw new Error('snapshot-harness: expected session/set_mode to be rejected but it succeeded') },
|
||||
() => { /* expected: the bridge rejected the mode id */ },
|
||||
)
|
||||
return
|
||||
}
|
||||
case 'setConfigOption': {
|
||||
const sessionId = getSessionId()
|
||||
if (sessionId === undefined) throw new Error('snapshot-harness: setConfigOption before newSession')
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
|
||||
export {
|
||||
runScenario,
|
||||
type ElicitationAnswer,
|
||||
type HarvestedLog,
|
||||
type InputScript,
|
||||
type InputStep,
|
||||
|
||||
@@ -15,6 +15,8 @@ import {
|
||||
ndJsonStream,
|
||||
type Agent as AcpAgent,
|
||||
type Client,
|
||||
type CreateElicitationRequest,
|
||||
type CreateElicitationResponse,
|
||||
type RequestPermissionRequest,
|
||||
type RequestPermissionResponse,
|
||||
type SessionNotification,
|
||||
@@ -47,6 +49,8 @@ export interface AcpTestLaunchOptions {
|
||||
env?: NodeJS.ProcessEnv
|
||||
/** Permission handler; omitted requests fail closed as `cancelled`. */
|
||||
requestPermission?: (params: RequestPermissionRequest) => Promise<RequestPermissionResponse>
|
||||
/** Elicitation handler; omitted requests fail closed as `cancel`. */
|
||||
createElicitation?: (params: CreateElicitationRequest) => Promise<CreateElicitationResponse>
|
||||
}
|
||||
|
||||
/** A running ACP test process and its captured client-side surfaces. */
|
||||
@@ -152,6 +156,8 @@ export function launchAcpTestAgent(options: AcpTestLaunchOptions): LaunchedAcpTe
|
||||
}
|
||||
const requestPermission = options.requestPermission
|
||||
?? (() => Promise.resolve({ outcome: { outcome: 'cancelled' as const } }))
|
||||
const createElicitation = options.createElicitation
|
||||
?? (() => Promise.resolve({ action: 'cancel' as const }))
|
||||
const makeClient = (_agent: AcpAgent): Client => ({
|
||||
sessionUpdate(params: SessionNotification): Promise<void> {
|
||||
return trackClientCallback(() => {
|
||||
@@ -175,6 +181,7 @@ export function launchAcpTestAgent(options: AcpTestLaunchOptions): LaunchedAcpTe
|
||||
})
|
||||
},
|
||||
requestPermission: params => trackClientCallback(() => requestPermission(params)),
|
||||
unstable_createElicitation: params => trackClientCallback(() => createElicitation(params)),
|
||||
})
|
||||
const client = new ClientSideConnection(makeClient, stream)
|
||||
// `exit` only reports the parent process's status. Descendants may retain
|
||||
|
||||
@@ -1,7 +1,19 @@
|
||||
/**
|
||||
* Scripted ACP agent for snapshot-kit tests. A fixture-adjacent `behavior.json` controls the
|
||||
* subprocess reached through the real harness path; the bin reports observations over ACP and
|
||||
* writes scripted logs before exiting on stdin EOF.
|
||||
* Scripted fake ACP agent bin for `dsh-acp-snapshot`'s unit specs. Speaks
|
||||
* newline-delimited JSON-RPC on stdio like the real `dsh-acp-agent` bin, but
|
||||
* every behavior — how prompts settle, whether session/new rejects, which
|
||||
* session logs get persisted, what filesystem noise to leave — comes from a
|
||||
* `behavior.json` sitting NEXT to the `$DSH_SNAPSHOT_FILE` fixture, so a spec
|
||||
* scripts a whole subprocess run from data. The specs launch it through the
|
||||
* REAL `runScenario` spawn path (tsx loader, temp cwd, env plumbing), so the
|
||||
* harness plumbing is exercised for real; only the agent behind the protocol
|
||||
* is scripted.
|
||||
*
|
||||
* The specs (not the golden tier) own this bin: it asserts nothing, echoes
|
||||
* observable facts into `session/update` text chunks (env probe, permission
|
||||
* outcome, seeded-workspace listing) for the spec to read off `rawStdout`, and
|
||||
* exits 0 on stdin EOF after writing the scripted logs — mirroring the real
|
||||
* bin's dispose-flush-exit shape.
|
||||
*/
|
||||
|
||||
import { mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
@@ -39,6 +51,10 @@ interface Behavior {
|
||||
cancelToolCallUpdate?: boolean
|
||||
/** Before responding to a prompt, send a `session/request_permission` request and echo its outcome as a chunk. */
|
||||
permissionProbe?: boolean
|
||||
/** Before responding to a prompt, send an `elicitation/create` request and echo its response as a chunk. */
|
||||
elicitationProbe?: boolean
|
||||
/** How `session/set_mode` settles: an empty response (echoing the modeId as a chunk) or a JSON-RPC error. */
|
||||
setMode?: 'respond' | 'error'
|
||||
/** Echo the `DSH_SNAPSHOT_*` env the harness set as a chunk (spec-side env-plumbing assertions). */
|
||||
echoEnv?: boolean
|
||||
/** Echo the sorted cwd listing as a chunk (spec-side workspace-seeding assertions). */
|
||||
@@ -84,8 +100,8 @@ let sessionId = ''
|
||||
let sessionCwd = ''
|
||||
/** The parked prompt request id while `hang-until-cancel` waits for the cancel notification. */
|
||||
let parkedPromptId: number | string | null = null
|
||||
/** Resolvers for permission-probe responses, keyed by outbound request id. */
|
||||
const pendingPermission = new Map<number, (outcome: unknown) => void>()
|
||||
/** Resolvers for outbound probe responses (permission/elicitation), keyed by request id. */
|
||||
const pendingOutbound = new Map<number, (result: unknown) => void>()
|
||||
/** Per-run `session/set_config_option` state: config id → current value (first vocabulary entry until set). */
|
||||
const currentConfig: Record<string, string> = {}
|
||||
|
||||
@@ -160,8 +176,8 @@ async function handlePrompt(id: number | string): Promise<void> {
|
||||
}
|
||||
if (behavior.permissionProbe === true) {
|
||||
const requestId = nextOutboundId++
|
||||
const outcome = await new Promise<unknown>((resolve) => {
|
||||
pendingPermission.set(requestId, resolve)
|
||||
const result = await new Promise<unknown>((resolve) => {
|
||||
pendingOutbound.set(requestId, resolve)
|
||||
send({
|
||||
id: requestId,
|
||||
method: 'session/request_permission',
|
||||
@@ -175,7 +191,24 @@ async function handlePrompt(id: number | string): Promise<void> {
|
||||
},
|
||||
})
|
||||
})
|
||||
chunk(`permission:${JSON.stringify(outcome)}`)
|
||||
chunk(`permission:${JSON.stringify((result as { outcome?: unknown } | undefined)?.outcome ?? null)}`)
|
||||
}
|
||||
if (behavior.elicitationProbe === true) {
|
||||
const requestId = nextOutboundId++
|
||||
const result = await new Promise<unknown>((resolve) => {
|
||||
pendingOutbound.set(requestId, resolve)
|
||||
send({
|
||||
id: requestId,
|
||||
method: 'elicitation/create',
|
||||
params: {
|
||||
sessionId,
|
||||
mode: 'form',
|
||||
message: 'Approve this plan and leave plan mode?',
|
||||
requestedSchema: { type: 'object', title: 'Plan review', properties: { choice: { type: 'string' }, custom: { type: 'string' } }, required: [] },
|
||||
},
|
||||
})
|
||||
})
|
||||
chunk(`elicitation:${JSON.stringify(result ?? null)}`)
|
||||
}
|
||||
switch (behavior.prompt ?? 'respond') {
|
||||
case 'respond':
|
||||
@@ -195,10 +228,10 @@ function handleFrame(frame: Record<string, unknown>): void {
|
||||
const method = frame.method as string | undefined
|
||||
const params = (frame.params ?? {}) as Record<string, unknown>
|
||||
// A response to one of OUR outbound requests (the permission probe).
|
||||
if (method === undefined && id !== undefined && typeof id === 'number' && pendingPermission.has(id)) {
|
||||
const resolve = pendingPermission.get(id) as (outcome: unknown) => void
|
||||
pendingPermission.delete(id)
|
||||
resolve((frame.result as { outcome?: unknown } | undefined)?.outcome ?? null)
|
||||
if (method === undefined && id !== undefined && typeof id === 'number' && pendingOutbound.has(id)) {
|
||||
const resolve = pendingOutbound.get(id) as (result: unknown) => void
|
||||
pendingOutbound.delete(id)
|
||||
resolve(frame.result)
|
||||
return
|
||||
}
|
||||
switch (method) {
|
||||
@@ -219,6 +252,14 @@ function handleFrame(frame: Record<string, unknown>): void {
|
||||
case 'session/prompt':
|
||||
void handlePrompt(id as number | string)
|
||||
return
|
||||
case 'session/set_mode':
|
||||
if ((behavior.setMode ?? 'respond') === 'error') {
|
||||
respondError(id as number | string, 'unknown mode')
|
||||
return
|
||||
}
|
||||
chunk(`setMode:${String(params.modeId)}`)
|
||||
respond(id as number | string, {})
|
||||
return
|
||||
case 'session/set_config_option': {
|
||||
const vocabulary = behavior.configOptions
|
||||
const configId = params.configId as string
|
||||
|
||||
2
packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/session.1.jsonl
vendored
Normal file
2
packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/session.1.jsonl
vendored
Normal file
@@ -0,0 +1,2 @@
|
||||
{"type":"session","id":"abababab-cdcd-4efe-8ada-badabadabada","createdAt":800,"cwd":"/var/folders/2g/b32ct0qn1d728l_v6tdkjytr0000gn/T/acp-snap-cwd-KBQJbW","parentSession":"f6fa7fcf-dd9c-4b39-8815-b25ddcebfd88"}
|
||||
{"type":"request/header","seq":0,"time":2,"data":{"header":{"config":{"model":"fake"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}}
|
||||
@@ -95,8 +95,8 @@ describe('runScenario', () => {
|
||||
expect(clientClosed).toBe(true)
|
||||
})
|
||||
|
||||
it('centralizes ACP boot, captures, updates, fail-closed permissions, and shutdown', { timeout: 20_000 }, async () => {
|
||||
const { dir, fixtureFile } = await scenario({ permissionProbe: true, echoEnv: true, stderrNote: 'launcher stderr' })
|
||||
it('centralizes ACP boot, captures, updates, fail-closed interactions, and shutdown', { timeout: 20_000 }, async () => {
|
||||
const { dir, fixtureFile } = await scenario({ permissionProbe: true, elicitationProbe: true, echoEnv: true, stderrNote: 'launcher stderr' })
|
||||
const sessionsRoot = await mkdtemp(join(tmpdir(), 'acp-launcher-sessions-'))
|
||||
tempDirs.push(sessionsRoot)
|
||||
const launched = launchAcpTestAgent({
|
||||
@@ -120,6 +120,7 @@ describe('runScenario', () => {
|
||||
expect((await nextChunk).sessionUpdate).toBe('agent_message_chunk')
|
||||
expect(launched.updates.some(update => update.sessionUpdate === 'agent_message_chunk')).toBe(true)
|
||||
expect(launched.rawStdout()).toContain('permission:{\\"outcome\\":\\"cancelled\\"}')
|
||||
expect(launched.rawStdout()).toContain('elicitation:{\\"action\\":\\"cancel\\"}')
|
||||
expect(launched.stderr()).toContain('launcher stderr')
|
||||
const unmatched = expect(launched.waitForUpdate(() => false)).rejects.toThrow(/update stream closed/)
|
||||
await launched.close()
|
||||
@@ -716,6 +717,69 @@ describe('runScenario', () => {
|
||||
expect(result.sessionLogs).toHaveLength(0)
|
||||
})
|
||||
|
||||
it('drives session/set_mode and swallows the expected rejection of setModeExpectError', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({})
|
||||
const result = await runScenario(
|
||||
{ steps: [...boot, { op: 'setMode', modeId: 'plan' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)
|
||||
expect(result.rawStdout).toContain('setMode:plan')
|
||||
|
||||
const rejecting = await scenario({ setMode: 'error' })
|
||||
const rejected = await runScenario(
|
||||
{ steps: [...boot, { op: 'setModeExpectError', modeId: 'yolo' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile: rejecting.fixtureFile },
|
||||
)
|
||||
expect(rejected.rawStdout).toContain('unknown mode')
|
||||
})
|
||||
|
||||
it('fails the run when setModeExpectError unexpectedly succeeds, and both mode ops require a session', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({})
|
||||
await expect(runScenario(
|
||||
{ steps: [...boot, { op: 'setModeExpectError', modeId: 'plan' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)).rejects.toThrow(/expected session\/set_mode to be rejected/)
|
||||
await expect(runScenario(
|
||||
{ steps: [{ op: 'initialize' }, { op: 'setMode', modeId: 'plan' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)).rejects.toThrow(/setMode before newSession/)
|
||||
await expect(runScenario(
|
||||
{ steps: [{ op: 'initialize' }, { op: 'setModeExpectError', modeId: 'plan' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)).rejects.toThrow(/setModeExpectError before newSession/)
|
||||
})
|
||||
|
||||
it('answers elicitations from the scripted queue, falling back to cancel on exhaustion', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ elicitationProbe: true })
|
||||
// Three prompts → three elicitations: an accept-with-choice, an
|
||||
// accept-with-custom (feedback), then the exhausted-queue cancel.
|
||||
const result = await runScenario(
|
||||
{
|
||||
steps: [...boot, { op: 'prompt', text: 'one' }, { op: 'prompt', text: 'two' }, { op: 'prompt', text: 'three' }],
|
||||
elicitationAnswers: [
|
||||
{ action: 'accept', choice: 'Approve' },
|
||||
{ action: 'accept', custom: 'add tests first' },
|
||||
],
|
||||
},
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)
|
||||
const first = result.rawStdout.indexOf('elicitation:{\\"action\\":\\"accept\\",\\"content\\":{\\"choice\\":\\"Approve\\"}}')
|
||||
const second = result.rawStdout.indexOf('elicitation:{\\"action\\":\\"accept\\",\\"content\\":{\\"custom\\":\\"add tests first\\"}}')
|
||||
const third = result.rawStdout.indexOf('elicitation:{\\"action\\":\\"cancel\\"}')
|
||||
expect(first).toBeGreaterThanOrEqual(0)
|
||||
expect(second).toBeGreaterThan(first)
|
||||
expect(third).toBeGreaterThan(second)
|
||||
})
|
||||
|
||||
it('a scripted elicitation cancel answers cancel', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ elicitationProbe: true })
|
||||
const result = await runScenario(
|
||||
{ steps: [...boot, { op: 'prompt', text: 'one' }], elicitationAnswers: [{ action: 'cancel' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)
|
||||
expect(result.rawStdout).toContain('elicitation:{\\"action\\":\\"cancel\\"}')
|
||||
})
|
||||
|
||||
it('answers permission requests from the scripted queue by option kind, falling back to cancelled', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ permissionProbe: true })
|
||||
// Two prompts → two permission round-trips; one scripted answer, so the
|
||||
@@ -744,8 +808,11 @@ describe('runScenario', () => {
|
||||
|
||||
it('rejects the run on a scripted permission kind the agent never offered', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ permissionProbe: true })
|
||||
// The fake offers only allow_once/reject_once. The harness must reject an impossible click,
|
||||
// not merely send an RPC error that a tolerant agent could absorb.
|
||||
// The fake bin offers allow_once/reject_once; scripting allow_always is a
|
||||
// scenario bug. The agent is answered `cancelled` (it must not be able to
|
||||
// absorb the bug as an error-means-denial), and the RUN fails: a callback
|
||||
// throw would only reach the agent as a JSON-RPC error response, letting
|
||||
// a tolerant agent carry on and the scenario pass — or record.
|
||||
await expect(runScenario(
|
||||
{ steps: [...boot, { op: 'prompt', text: 'impossible click' }], permissionAnswers: [{ kind: 'allow_always' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
|
||||
Reference in New Issue
Block a user