Merge remote-tracking branch 'origin/master' into worktree-windows-runtime
# Conflicts: # packages/session-persistence/session-persistence-jsonl/README.md # packages/session-persistence/session-persistence-jsonl/src/index.ts # packages/session-persistence/session-persistence-jsonl/tests/jsonl.spec.ts # packages/support/loader-smoke/tests/loader-smoke.spec.ts
This commit is contained in:
@@ -37,11 +37,10 @@ export type { AgentUnderTest } from './launcher.ts'
|
||||
* (random) session id into a `{{sessionId}}` variable that later steps
|
||||
* reference, since a committed file cannot know the id in advance.
|
||||
*
|
||||
* `promptAndCancel` sends a prompt WITHOUT awaiting its response, waits until
|
||||
* the client observes the first streamed `agent_message_chunk` (so the emitted
|
||||
* frames deterministically precede the cancellation), then cancels the turn —
|
||||
* the only way to exercise a cancel deterministically (a plain `prompt` step
|
||||
* awaits the response, which a cancel/hang scenario would block on forever).
|
||||
* `promptAndCancel` starts a prompt without awaiting completion, waits until
|
||||
* the client observes the selected update (`agent_message_chunk` by default),
|
||||
* then cancels and awaits completion. A named `waitForToolCallUpdate` keeps the
|
||||
* step open for a terminal tool update that may follow the prompt response.
|
||||
*/
|
||||
export type InputStep =
|
||||
| { op: 'initialize'; terminalOutput?: boolean }
|
||||
@@ -49,7 +48,12 @@ export type InputStep =
|
||||
| { op: 'newSessionExpectError'; additionalDirectories?: string[] }
|
||||
| { op: 'prompt'; text: string }
|
||||
| { op: 'promptExpectError'; text: string }
|
||||
| { op: 'promptAndCancel'; text: string }
|
||||
| {
|
||||
op: 'promptAndCancel'
|
||||
text: string
|
||||
afterUpdate?: 'agent_message_chunk' | 'tool_call'
|
||||
waitForToolCallUpdate?: string
|
||||
}
|
||||
| { op: 'cancel' }
|
||||
| { op: 'setConfigOption'; configId: string; value: string }
|
||||
| { op: 'setConfigOptionExpectError'; configId: string; value: string }
|
||||
@@ -351,17 +355,19 @@ async function runStep(
|
||||
case 'promptAndCancel': {
|
||||
const sessionId = getSessionId()
|
||||
if (sessionId === undefined) throw new Error('snapshot-harness: promptAndCancel before newSession')
|
||||
// Dispatch the prompt WITHOUT awaiting (a hang fixture never resolves on
|
||||
// its own). To pin frame order deterministically, wait until the client
|
||||
// has OBSERVED the hang's streamed agent_message_chunk before cancelling —
|
||||
// so those update frames always precede the cancelled prompt response in
|
||||
// the transcript (without this, the late chunk and the response race).
|
||||
// Then cancel and await the prompt, which the bridge settles as
|
||||
// `cancelled` once the abort propagates.
|
||||
// Dispatch without awaiting because the fixture does not settle on its
|
||||
// own. Waiting for the selected update pins it before cancellation and
|
||||
// the cancelled prompt response in the transcript.
|
||||
const promptDone = client.prompt({ sessionId, prompt: [{ type: 'text', text: step.text }] })
|
||||
await waitForUpdate(u => u.sessionUpdate === 'agent_message_chunk')
|
||||
const afterUpdate = step.afterUpdate ?? 'agent_message_chunk'
|
||||
await waitForUpdate(u => u.sessionUpdate === afterUpdate)
|
||||
// Arm this before cancellation so a fast tool drain cannot outrun the waiter.
|
||||
const toolCallUpdateDone = step.waitForToolCallUpdate === undefined
|
||||
? undefined
|
||||
: waitForUpdate(u => u.sessionUpdate === 'tool_call_update' && u.toolCallId === step.waitForToolCallUpdate)
|
||||
await client.cancel({ sessionId })
|
||||
await promptDone
|
||||
if (toolCallUpdateDone !== undefined) await toolCallUpdateDone
|
||||
return
|
||||
}
|
||||
case 'cancel': {
|
||||
|
||||
@@ -33,6 +33,10 @@ interface Behavior {
|
||||
rejectExtraDirs?: boolean
|
||||
/** How `session/prompt` settles: a clean response, a JSON-RPC error, or a hang until `session/cancel`. */
|
||||
prompt?: 'respond' | 'error' | 'hang-until-cancel'
|
||||
/** Emit a tool call instead of a message chunk before parking a cancellable prompt. */
|
||||
cancelAtToolCall?: boolean
|
||||
/** Emit the parked tool call's terminal update after answering cancellation. */
|
||||
cancelToolCallUpdate?: boolean
|
||||
/** Before responding to a prompt, send a `session/request_permission` request and echo its outcome as a chunk. */
|
||||
permissionProbe?: boolean
|
||||
/** Echo the `DSH_SNAPSHOT_*` env the harness set as a chunk (spec-side env-plumbing assertions). */
|
||||
@@ -126,7 +130,23 @@ async function handlePrompt(id: number | string): Promise<void> {
|
||||
params: { sessionId, update: { sessionUpdate: 'agent_thought_chunk', content: { type: 'text', text: 'mulling' } } },
|
||||
})
|
||||
}
|
||||
chunk('thinking about it')
|
||||
if (behavior.cancelAtToolCall === true) {
|
||||
send({
|
||||
method: 'session/update',
|
||||
params: {
|
||||
sessionId,
|
||||
update: {
|
||||
sessionUpdate: 'tool_call',
|
||||
toolCallId: 'call_fake_1',
|
||||
title: 'fake tool',
|
||||
kind: 'execute',
|
||||
status: 'in_progress',
|
||||
},
|
||||
},
|
||||
})
|
||||
} else {
|
||||
chunk('thinking about it')
|
||||
}
|
||||
if (behavior.echoEnv === true) {
|
||||
chunk(`env:${JSON.stringify({
|
||||
mode: process.env.DSH_SNAPSHOT,
|
||||
@@ -229,6 +249,19 @@ function handleFrame(frame: Record<string, unknown>): void {
|
||||
const parked = parkedPromptId
|
||||
parkedPromptId = null
|
||||
respond(parked, { stopReason: 'cancelled' })
|
||||
if (behavior.cancelToolCallUpdate === true) {
|
||||
send({
|
||||
method: 'session/update',
|
||||
params: {
|
||||
sessionId,
|
||||
update: {
|
||||
sessionUpdate: 'tool_call_update',
|
||||
toolCallId: 'call_fake_1',
|
||||
status: 'failed',
|
||||
},
|
||||
},
|
||||
})
|
||||
}
|
||||
}
|
||||
return
|
||||
default:
|
||||
|
||||
@@ -447,6 +447,28 @@ describe('runScenario', () => {
|
||||
expect(result.rawStdout.indexOf('thinking about it')).toBeLessThan(result.rawStdout.indexOf('cancelled'))
|
||||
})
|
||||
|
||||
it('promptAndCancel can bracket cancellation with tool-call updates', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({
|
||||
prompt: 'hang-until-cancel',
|
||||
cancelAtToolCall: true,
|
||||
cancelToolCallUpdate: true,
|
||||
})
|
||||
const result = await runScenario(
|
||||
{
|
||||
steps: [...boot, {
|
||||
op: 'promptAndCancel',
|
||||
text: 'hang',
|
||||
afterUpdate: 'tool_call',
|
||||
waitForToolCallUpdate: 'call_fake_1',
|
||||
}],
|
||||
},
|
||||
{ agent: AGENT, mode: 'replay', fixtureFile },
|
||||
)
|
||||
expect(result.rawStdout).toContain('"sessionUpdate":"tool_call"')
|
||||
expect(result.rawStdout.indexOf('"sessionUpdate":"tool_call"')).toBeLessThan(result.rawStdout.indexOf('cancelled'))
|
||||
expect(result.rawStdout.indexOf('cancelled')).toBeLessThan(result.rawStdout.indexOf('"sessionUpdate":"tool_call_update"'))
|
||||
})
|
||||
|
||||
it('promptExpectError swallows a model-error response as the expected outcome', { timeout: 20_000 }, async () => {
|
||||
const { fixtureFile } = await scenario({ prompt: 'error' })
|
||||
const result = await runScenario(
|
||||
|
||||
@@ -30,10 +30,12 @@ const scopedSubjectResolvers = Object.freeze({
|
||||
'agent/created': adapt<'agent/created'>(args => args[0]),
|
||||
'agent/disposed': adapt<'agent/disposed'>(args => args[0]),
|
||||
'agent/error': adapt<'agent/error'>(args => args[0]),
|
||||
'agent/post-step': adapt<'agent/post-step'>(args => args[0]),
|
||||
'agent/pre-step': adapt<'agent/pre-step'>(args => args[0]),
|
||||
'agent/prompt-submit': adapt<'agent/prompt-submit'>(args => args[0]),
|
||||
'agent/queued': adapt<'agent/queued'>(args => args[0]),
|
||||
'agent/request': adapt<'agent/request'>(args => args[0]),
|
||||
'agent/request-error': adapt<'agent/request-error'>(args => args[0]),
|
||||
'agent/session-prefix': adapt<'agent/session-prefix'>(args => args[0]),
|
||||
'agent/session-start': adapt<'agent/session-start'>(args => args[0]),
|
||||
'agent/status': adapt<'agent/status'>(args => args[0]),
|
||||
|
||||
@@ -814,7 +814,7 @@ describe('scoped-dispatch invariants', () => {
|
||||
['agent/status', [agent, 'idle']],
|
||||
['agent/queued', [agent, [], { source: { kind: 'user' }, steering: false }]],
|
||||
['agent/session-start', [agent, 'startup']],
|
||||
['agent/pre-step', [agent, 1, 1, '', new AbortController().signal]],
|
||||
['agent/pre-step', [agent, 1, 1, new AbortController().signal]],
|
||||
['agent/prompt-submit', [agent, [], { kind: 'user' }, () => Promise.resolve({ kind: 'allow' })]],
|
||||
['agent/request', [agent, 1, 1, { model: 'm' }, () => Promise.resolve({ model: 'm' })]],
|
||||
['agent/session-prefix', [agent, [], new AbortController().signal, () => Promise.resolve([])]],
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
A replay LLM plugin for keyless snapshot tests. It yields model streams reconstructed from a recorded **session JSONL** fixture, so a test can boot the real agent against a fixed model transcript with no API key. With `providers` configured it registers a replay-only adapter whose catalog is visible to clients such as ACP editors; without `providers` it installs the catch-all `llm/stream` waterfall used by tests that do not need discovery.
|
||||
|
||||
Its consumer is the ACP snapshot harness in `examples/acp-agent`, which loads this plugin (via `cordis.snapshot.yml`) in place of a real LLM adapter. The package exists so its derive/parse/replay logic falls under the per-file 100% coverage gate on `packages/*/src` (the same logic, while it lived under `examples/`, was outside the gate).
|
||||
Its consumers are the ACP snapshot harness in `examples/acp-agent` and the `stream-json` snapshot in `examples/headless-agent`; each loads this plugin in place of a real LLM adapter. Keeping derivation and replay here places that logic under the per-file 100% coverage gate on `packages/*/src`.
|
||||
|
||||
## How the fixture works
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
Shared subprocess harness for tests that boot an app and `cordis.yml` through the Cordis Loader. `resolveExampleLaunch` selects local `src` mode (tsx and root tsconfig paths) or CI `lib` mode (plain Node and package exports) from an explicit mode or `DSH_EXAMPLE_MODE`.
|
||||
|
||||
`runLoaderSmoke` owns the isolated cwd, DSH homes, stdin, diagnostics, deadline, termination, and cleanup. It returns both streams after a zero exit and rejects with both streams on failure.
|
||||
`runLoaderSmoke` accepts bin and config paths, optional complete bin arguments, environment overrides, stdin, pre-run setup, and pre-cleanup inspection. It owns the isolated cwd, DSH homes, diagnostics, deadline, termination, EOF, and cleanup; it returns both streams after a zero exit and rejects with both streams on failure.
|
||||
|
||||
This is support-tier test infrastructure, not product API.
|
||||
|
||||
|
||||
@@ -1,14 +1,12 @@
|
||||
/**
|
||||
* Shared subprocess harness for keyless example smokes that boot a real
|
||||
* `cordis.yml` through the stdio-agent bin and Cordis Loader.
|
||||
* `cordis.yml` through an app bin and Cordis Loader.
|
||||
*
|
||||
* It also owns the mode-aware launch resolver every example subprocess harness shares
|
||||
* ({@link resolveExampleLaunch}): booting an example bin from TypeScript source under `tsx` (the
|
||||
* zero-build dev path, resolving `@deepseek-ai/dsh-*` / `@cordisjs/*` through the tsconfig `paths`
|
||||
* map) or from built `lib/` under plain Node (resolving bare packages through real `exports`, as an
|
||||
* installed consumer does, while Node type-strips relative example-local TypeScript plugins).
|
||||
* Consolidating that spawn glue here retires the copies in the ACP snapshot harness and the example
|
||||
* e2e drivers (the `TODO(acp-test-harness)`).
|
||||
*
|
||||
* @module @deepseek-ai/dsh-loader-smoke
|
||||
*/
|
||||
@@ -126,12 +124,14 @@ export interface LoaderSmokeOptions {
|
||||
readonly label: string
|
||||
/** Prefix for the isolated temporary process cwd. */
|
||||
readonly tempDirPrefix: string
|
||||
/** Absolute stdio-agent bin SOURCE path (`<pkg>/src/bin.ts`); the `lib` bin is derived from it. */
|
||||
/** Absolute app-bin source path (`<pkg>/src/bin.ts`); the `lib` bin is derived from it. */
|
||||
readonly binScript: string
|
||||
/** Explicit plain-Node entry for `lib` mode; intended for test fixtures outside a package `src/` tree. */
|
||||
readonly libBinScript?: string | undefined
|
||||
/** Absolute real Loader config path. */
|
||||
/** Absolute real Loader config path, passed as the sole bin argument by default. */
|
||||
readonly configPath: string
|
||||
/** Complete argv after the bin path; overrides the default `[configPath]`. */
|
||||
readonly binArgs?: readonly string[]
|
||||
/** Absolute repo tsconfig path used for unbuilt workspace-package resolution (required in `src` mode). */
|
||||
readonly tsconfigPath: string
|
||||
/** Boot from source via tsx (`src`) or built lib via plain Node (`lib`); defaults to the environment's mode. */
|
||||
@@ -142,6 +142,10 @@ export interface LoaderSmokeOptions {
|
||||
readonly stdinLines?: readonly string[]
|
||||
/** Process deadline override for harness tests. */
|
||||
readonly processTimeoutMs?: number
|
||||
/** Optional world-state setup run in the isolated cwd before process start. */
|
||||
readonly prepare?: (cwd: string) => Promise<void> | void
|
||||
/** Optional world-state assertion run in the isolated cwd before cleanup. */
|
||||
readonly inspect?: (cwd: string) => Promise<void> | void
|
||||
}
|
||||
|
||||
/** Captured output from a Loader smoke that exited successfully. */
|
||||
@@ -162,17 +166,18 @@ export interface LoaderSmokeResult {
|
||||
export async function runLoaderSmoke(options: LoaderSmokeOptions): Promise<LoaderSmokeResult> {
|
||||
const cwd = await mkdtemp(join(tmpdir(), options.tempDirPrefix))
|
||||
const processTimeoutMs = options.processTimeoutMs ?? DEFAULT_PROCESS_TIMEOUT_MS
|
||||
const launch = resolveExampleLaunch({
|
||||
srcBin: options.binScript,
|
||||
libBin: options.libBinScript,
|
||||
configArgs: [options.configPath],
|
||||
...options.mode !== undefined ? { mode: options.mode } : {},
|
||||
tsconfigPath: options.tsconfigPath,
|
||||
exposeInternals: true,
|
||||
env: { DSH_HOME: join(cwd, '.dsh'), DSH_AGENTS_HOME: join(cwd, '.agents'), ...options.env },
|
||||
})
|
||||
try {
|
||||
return await new Promise((resolve, reject) => {
|
||||
await options.prepare?.(cwd)
|
||||
const launch = resolveExampleLaunch({
|
||||
srcBin: options.binScript,
|
||||
libBin: options.libBinScript,
|
||||
configArgs: options.binArgs ?? [options.configPath],
|
||||
...options.mode !== undefined ? { mode: options.mode } : {},
|
||||
tsconfigPath: options.tsconfigPath,
|
||||
exposeInternals: true,
|
||||
env: { DSH_HOME: join(cwd, '.dsh'), DSH_AGENTS_HOME: join(cwd, '.agents'), ...options.env },
|
||||
})
|
||||
const result = await new Promise<LoaderSmokeResult>((resolve, reject) => {
|
||||
const child = spawn(launch.command, launch.args, {
|
||||
cwd,
|
||||
env: { ...process.env, ...launch.env },
|
||||
@@ -217,6 +222,8 @@ export async function runLoaderSmoke(options: LoaderSmokeOptions): Promise<Loade
|
||||
|
||||
child.stdin.end((options.stdinLines ?? []).map(line => `${line}\n`).join(''))
|
||||
})
|
||||
await options.inspect?.(cwd)
|
||||
return result
|
||||
} finally {
|
||||
await rm(cwd, { recursive: true, force: true })
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ process.stdin.on('data', (chunk: string) => { input += chunk })
|
||||
process.stdin.on('end', () => {
|
||||
console.log(JSON.stringify({
|
||||
configPath: process.argv[2],
|
||||
args: process.argv.slice(2),
|
||||
cwd: process.cwd(),
|
||||
dshHome: process.env.DSH_HOME,
|
||||
agentsHome: process.env.DSH_AGENTS_HOME,
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { existsSync } from 'node:fs'
|
||||
import { readFile, writeFile } from 'node:fs/promises'
|
||||
import { join } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import { describe, expect, it } from 'vitest'
|
||||
@@ -23,6 +24,7 @@ describe('runLoaderSmoke', () => {
|
||||
})
|
||||
const output = JSON.parse(result.stdout) as {
|
||||
configPath: string
|
||||
args: string[]
|
||||
cwd: string
|
||||
dshHome: string
|
||||
agentsHome: string
|
||||
@@ -31,6 +33,7 @@ describe('runLoaderSmoke', () => {
|
||||
}
|
||||
expect(output).toMatchObject({
|
||||
configPath,
|
||||
args: [configPath],
|
||||
marker: 'present',
|
||||
input: 'one\ntwo\n',
|
||||
})
|
||||
@@ -40,6 +43,30 @@ describe('runLoaderSmoke', () => {
|
||||
expect(existsSync(output.cwd)).toBe(false)
|
||||
}, LOADER_SMOKE_TEST_TIMEOUT_MS)
|
||||
|
||||
it('passes an arbitrary bin argv and inspects world state before cleanup', async () => {
|
||||
let inspected = ''
|
||||
let marker = ''
|
||||
const result = await runLoaderSmoke({
|
||||
label: 'argv fixture',
|
||||
tempDirPrefix: 'loader-smoke-argv-',
|
||||
binScript: fixture('success'),
|
||||
libBinScript: fixture('success'),
|
||||
configPath,
|
||||
binArgs: ['--config', configPath, '--output-format', 'json', 'task with spaces'],
|
||||
tsconfigPath,
|
||||
prepare: cwd => writeFile(join(cwd, 'marker.txt'), 'prepared'),
|
||||
inspect: async (cwd) => {
|
||||
inspected = cwd
|
||||
marker = await readFile(join(cwd, 'marker.txt'), 'utf8')
|
||||
},
|
||||
})
|
||||
const output = JSON.parse(result.stdout) as { args: string[]; cwd: string }
|
||||
expect(output.args).toEqual(['--config', configPath, '--output-format', 'json', 'task with spaces'])
|
||||
expect(canonicalTempPath(inspected)).toBe(canonicalTempPath(output.cwd))
|
||||
expect(marker).toBe('prepared')
|
||||
expect(existsSync(inspected)).toBe(false)
|
||||
}, LOADER_SMOKE_TEST_TIMEOUT_MS)
|
||||
|
||||
it('rejects a non-zero exit with captured diagnostics', async () => {
|
||||
await expect(runLoaderSmoke({
|
||||
label: 'failure fixture',
|
||||
|
||||
Reference in New Issue
Block a user