test(acp): snapshot cancelled queued tool calls
This commit is contained in:
@@ -37,11 +37,10 @@ export type { AgentUnderTest } from './launcher.ts'
|
||||
* (random) session id into a `{{sessionId}}` variable that later steps
|
||||
* reference, since a committed file cannot know the id in advance.
|
||||
*
|
||||
* `promptAndCancel` sends a prompt WITHOUT awaiting its response, waits until
|
||||
* the client observes the first streamed `agent_message_chunk` (so the emitted
|
||||
* frames deterministically precede the cancellation), then cancels the turn —
|
||||
* the only way to exercise a cancel deterministically (a plain `prompt` step
|
||||
* awaits the response, which a cancel/hang scenario would block on forever).
|
||||
* `promptAndCancel` starts a prompt without awaiting completion, waits until
|
||||
* the client observes the selected update (`agent_message_chunk` by default),
|
||||
* then cancels and awaits completion. This keeps update/cancel order
|
||||
* deterministic for fixtures that a plain `prompt` cannot drive.
|
||||
*/
|
||||
export type InputStep =
|
||||
| { op: 'initialize'; terminalOutput?: boolean }
|
||||
@@ -49,7 +48,7 @@ export type InputStep =
|
||||
| { op: 'newSessionExpectError'; additionalDirectories?: string[] }
|
||||
| { op: 'prompt'; text: string }
|
||||
| { op: 'promptExpectError'; text: string }
|
||||
| { op: 'promptAndCancel'; text: string }
|
||||
| { op: 'promptAndCancel'; text: string; afterUpdate?: 'agent_message_chunk' | 'tool_call' }
|
||||
| { op: 'cancel' }
|
||||
| { op: 'setConfigOption'; configId: string; value: string }
|
||||
| { op: 'setConfigOptionExpectError'; configId: string; value: string }
|
||||
@@ -342,15 +341,12 @@ async function runStep(
|
||||
case 'promptAndCancel': {
|
||||
const sessionId = getSessionId()
|
||||
if (sessionId === undefined) throw new Error('snapshot-harness: promptAndCancel before newSession')
|
||||
// Dispatch the prompt WITHOUT awaiting (a hang fixture never resolves on
|
||||
// its own). To pin frame order deterministically, wait until the client
|
||||
// has OBSERVED the hang's streamed agent_message_chunk before cancelling —
|
||||
// so those update frames always precede the cancelled prompt response in
|
||||
// the transcript (without this, the late chunk and the response race).
|
||||
// Then cancel and await the prompt, which the bridge settles as
|
||||
// `cancelled` once the abort propagates.
|
||||
// Dispatch without awaiting because the fixture does not settle on its
|
||||
// own. Waiting for the selected update pins it before cancellation and
|
||||
// the cancelled prompt response in the transcript.
|
||||
const promptDone = client.prompt({ sessionId, prompt: [{ type: 'text', text: step.text }] })
|
||||
await waitForUpdate(u => u.sessionUpdate === 'agent_message_chunk')
|
||||
const afterUpdate = step.afterUpdate ?? 'agent_message_chunk'
|
||||
await waitForUpdate(u => u.sessionUpdate === afterUpdate)
|
||||
await client.cancel({ sessionId })
|
||||
await promptDone
|
||||
return
|
||||
|
||||
@@ -33,6 +33,8 @@ interface Behavior {
|
||||
rejectExtraDirs?: boolean
|
||||
/** How `session/prompt` settles: a clean response, a JSON-RPC error, or a hang until `session/cancel`. */
|
||||
prompt?: 'respond' | 'error' | 'hang-until-cancel'
|
||||
/** Emit a tool call instead of a message chunk before parking a cancellable prompt. */
|
||||
cancelAtToolCall?: boolean
|
||||
/** Before responding to a prompt, send a `session/request_permission` request and echo its outcome as a chunk. */
|
||||
permissionProbe?: boolean
|
||||
/** Echo the `DSH_SNAPSHOT_*` env the harness set as a chunk (spec-side env-plumbing assertions). */
|
||||
@@ -126,7 +128,23 @@ async function handlePrompt(id: number | string): Promise<void> {
|
||||
params: { sessionId, update: { sessionUpdate: 'agent_thought_chunk', content: { type: 'text', text: 'mulling' } } },
|
||||
})
|
||||
}
|
||||
chunk('thinking about it')
|
||||
if (behavior.cancelAtToolCall === true) {
|
||||
send({
|
||||
method: 'session/update',
|
||||
params: {
|
||||
sessionId,
|
||||
update: {
|
||||
sessionUpdate: 'tool_call',
|
||||
toolCallId: 'call_fake_1',
|
||||
title: 'fake tool',
|
||||
kind: 'execute',
|
||||
status: 'in_progress',
|
||||
},
|
||||
},
|
||||
})
|
||||
} else {
|
||||
chunk('thinking about it')
|
||||
}
|
||||
if (behavior.echoEnv === true) {
|
||||
chunk(`env:${JSON.stringify({
|
||||
mode: process.env.DSH_SNAPSHOT,
|
||||
|
||||
@@ -334,6 +334,16 @@ describe('runScenario', () => {
|
||||
expect(result.rawStdout.indexOf('thinking about it')).toBeLessThan(result.rawStdout.indexOf('cancelled'))
|
||||
})
|
||||
|
||||
it('promptAndCancel can wait for a tool call before cancelling', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ prompt: 'hang-until-cancel', cancelAtToolCall: true })
|
||||
const result = await runScenario(
|
||||
{ steps: [...boot, { op: 'promptAndCancel', text: 'hang', afterUpdate: 'tool_call' }] },
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)
|
||||
expect(result.rawStdout).toContain('"sessionUpdate":"tool_call"')
|
||||
expect(result.rawStdout.indexOf('"sessionUpdate":"tool_call"')).toBeLessThan(result.rawStdout.indexOf('cancelled'))
|
||||
})
|
||||
|
||||
it('promptExpectError swallows a model-error response as the expected outcome', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ prompt: 'error' })
|
||||
const result = await runScenario(
|
||||
|
||||
Reference in New Issue
Block a user