From 74502fa8c2e92bc85f26ec1ec7d259282609da8e Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Mon, 6 Jul 2026 23:29:08 +0800 Subject: [PATCH 01/24] Structured output on the subagent seam: schema subset, capture runtime, spawn/fork support MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Carved out of #170 per review feedback — the foundation the workflow tool builds on, now standing alone on master: - dsh-tools: the structured-output JSON Schema subset (StructuredOutputSchema, assertSupportedOutputSchema, validateStructuredValue) — rejects loud outside the enforced subset, listing every violation - dsh-subagent: SubagentStartRequest.outputSchema / SubagentResult.structured become a real capability; the service rejects a schema'd request whose provider lacks it - dsh-subagent-inprocess: the shared structured runtime — one global structured_output capture tool, a prepend final-assembly listener that strips the placeholder for plain agents and swaps in the run's own schema (plus the calling instruction as a trailing section) for structured children, an agent/turn-continuation veto once captured, and the capture/nudge loop in the run driver (structuredNudgeRetries, cancellation honored mid-nudge); lifetime refcounted by backends and live runs - subagent-spawn / subagent-fork flip outputSchema: true One deliberate divergence from the #170 revision: the backends do NOT add 'tools' to their plugin inject. Doing so deferred their apply past the todo plugin, and the delegation tool mirrors provider lifecycle — so the model-visible tool order of every existing prompt changed, invalidating every recorded snapshot fixture. The runtime now gates its capture-tool registration on tools availability itself (sync when live, a scoped inject fiber when the Loader starts the backend first), keeping this PR byte-invisible to existing transcripts: all 35 snapshot scenarios pass against master's fixtures unchanged. --- docs/cordis-catalog/events.md | 6 +- docs/cordis-catalog/services.md | 2 +- docs/core-data-structures/subagent.md | 4 +- docs/event-producer-consumer.md | 6 +- docs/module-graph.md | 4 +- packages/core/tools/README.md | 6 + packages/core/tools/src/index.ts | 10 + packages/core/tools/src/json-schema.ts | 322 ++++++++++++ packages/core/tools/tests/json-schema.spec.ts | 254 +++++++++ packages/subagent/subagent-fork/README.md | 3 +- packages/subagent/subagent-fork/src/index.ts | 37 +- .../tests/multi-subagent.spec.ts | 4 +- .../subagent-fork/tests/subagent-fork.spec.ts | 14 +- .../subagent/subagent-inprocess/README.md | 21 +- .../subagent/subagent-inprocess/package.json | 4 + .../subagent/subagent-inprocess/src/index.ts | 99 +++- .../subagent-inprocess/src/structured.ts | 239 +++++++++ .../tests/structured.spec.ts | 485 ++++++++++++++++++ .../tests/subagent-inprocess.spec.ts | 6 +- .../subagent/subagent-inprocess/tsconfig.json | 6 + packages/subagent/subagent-spawn/README.md | 5 +- packages/subagent/subagent-spawn/src/index.ts | 51 +- .../subagent/subagent-spawn/tests/harness.ts | 2 +- .../tests/subagent-spawn.spec.ts | 14 +- packages/subagent/subagent/src/types.ts | 14 +- .../subagent/subagent/tests/service.spec.ts | 4 +- .../subagent-mock/tests/subagent-mock.spec.ts | 4 +- pnpm-lock.yaml | 6 + 28 files changed, 1570 insertions(+), 62 deletions(-) create mode 100644 packages/core/tools/src/json-schema.ts create mode 100644 packages/core/tools/tests/json-schema.spec.ts create mode 100644 packages/subagent/subagent-inprocess/src/structured.ts create mode 100644 packages/subagent/subagent-inprocess/tests/structured.spec.ts diff --git a/docs/cordis-catalog/events.md b/docs/cordis-catalog/events.md index 753cf8f4d0..0f7ed1dc68 100644 --- a/docs/cordis-catalog/events.md +++ b/docs/cordis-catalog/events.md @@ -307,7 +307,7 @@ A tool was registered or unregistered (the available tool set changed). 'tools/change'(): void ``` -Source: [`packages/core/tools/src/index.ts:87`](../../packages/core/tools/src/index.ts) +Source: [`packages/core/tools/src/index.ts:97`](../../packages/core/tools/src/index.ts) ### `tools/post-execute` — waterfall @@ -319,7 +319,7 @@ Waterfall AFTER a tool runs — where hook plugins inspect the result and accept Types: [ToolExecution](../core-data-structures/tools.md) · [ToolExecutionResult](../core-data-structures/tools.md) -Source: [`packages/core/tools/src/index.ts:82`](../../packages/core/tools/src/index.ts) +Source: [`packages/core/tools/src/index.ts:92`](../../packages/core/tools/src/index.ts) ### `tools/pre-execute` — waterfall @@ -331,7 +331,7 @@ Waterfall BEFORE a tool runs — the gate where sandbox, permission, and hook pl Types: [ToolExecution](../core-data-structures/tools.md) -Source: [`packages/core/tools/src/index.ts:66`](../../packages/core/tools/src/index.ts) +Source: [`packages/core/tools/src/index.ts:76`](../../packages/core/tools/src/index.ts) ## Inherited events (cordis core + loader/hmr/timer) diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 0221946118..58186533fd 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -204,7 +204,7 @@ async execute(exec: ToolExecution): Promise Types: [ToolDefinition](../core-data-structures/tools.md) · [ToolExecution](../core-data-structures/tools.md) · [ToolExecutionResult](../core-data-structures/tools.md) -Source: [`packages/core/tools/src/index.ts:268`](../../packages/core/tools/src/index.ts) +Source: [`packages/core/tools/src/index.ts:278`](../../packages/core/tools/src/index.ts) ## `ctx.web` — `WebService` diff --git a/docs/core-data-structures/subagent.md b/docs/core-data-structures/subagent.md index 926e6ce9b8..f21ef15669 100644 --- a/docs/core-data-structures/subagent.md +++ b/docs/core-data-structures/subagent.md @@ -20,7 +20,7 @@ interface SubagentCapabilities { ## The start request -What a caller asks for when starting a subagent. The tool layer builds this from the model's `{ description, prompt }` plus its own config; the service validates the start-time capabilities against the named provider, then passes it to `provider.start`. `parent` is REQUIRED — in-process backends read `parent.session.header` for the working directory, the `parentSession` lineage, and the delegation depth. The three optional fields (`outputSchema`, `maxDepth`, `toolFilter`) each gate on the matching `SubagentCapabilities` flag. +What a caller asks for when starting a subagent. The tool layer builds this from the model's `{ description, prompt }` plus its own config; the service validates the start-time capabilities against the named provider, then passes it to `provider.start`. `parent` is REQUIRED — in-process backends read `parent.session.header` for the working directory, the `parentSession` lineage, and the delegation depth. The three optional fields (`outputSchema`, `maxDepth`, `toolFilter`) each gate on the matching `SubagentCapabilities` flag. `outputSchema` is an object-rooted JSON Schema within the subset `assertSupportedOutputSchema` (dsh-tools) enforces — a schema outside it is rejected loud at start; the in-process backends realize it with a forced `structured_output` capture tool (see the [driver README](../../packages/subagent/subagent-inprocess/README.md)). ```ts type-equiv interface SubagentStartRequest { @@ -28,7 +28,7 @@ interface SubagentStartRequest { parent: Agent signal?: AbortSignal agentOptions?: AgentOptions - outputSchema?: SchemaSpec + outputSchema?: StructuredOutputSchema maxDepth?: number toolFilter?: { allow?: string[]; deny?: string[] } } diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index 24a2793826..a77015ceda 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -31,8 +31,8 @@ This matrix shows which packages dispatch each harness-owned event and which pac | `subagent/start` | `emit` | [`packages/subagent/subagent/src/index.ts:91`](../packages/subagent/subagent/src/index.ts) | [`subagent`](../packages/subagent/subagent) (`events.dispatch`) | [`hooks-claude`](../packages/hooks/hooks-claude) | | `system-prompt/assemble` | `waterfall` | [`packages/core/system-prompt/src/index.ts:38`](../packages/core/system-prompt/src/index.ts) | [`system-prompt`](../packages/core/system-prompt) (`waterfall`) | - | | `system-prompt/change` | `emit` | [`packages/core/system-prompt/src/index.ts:44`](../packages/core/system-prompt/src/index.ts) | [`system-prompt`](../packages/core/system-prompt) (`emit`) | - | -| `tools/change` | `emit` | [`packages/core/tools/src/index.ts:87`](../packages/core/tools/src/index.ts) | [`tools`](../packages/core/tools) (`emit`) | - | -| `tools/post-execute` | `waterfall` | [`packages/core/tools/src/index.ts:82`](../packages/core/tools/src/index.ts) | [`tools`](../packages/core/tools) (`waterfall`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | -| `tools/pre-execute` | `waterfall` | [`packages/core/tools/src/index.ts:66`](../packages/core/tools/src/index.ts) | [`tools`](../packages/core/tools) (`waterfall`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | +| `tools/change` | `emit` | [`packages/core/tools/src/index.ts:97`](../packages/core/tools/src/index.ts) | [`tools`](../packages/core/tools) (`emit`) | - | +| `tools/post-execute` | `waterfall` | [`packages/core/tools/src/index.ts:92`](../packages/core/tools/src/index.ts) | [`tools`](../packages/core/tools) (`waterfall`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | +| `tools/pre-execute` | `waterfall` | [`packages/core/tools/src/index.ts:76`](../packages/core/tools/src/index.ts) | [`tools`](../packages/core/tools) (`waterfall`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`hooks-codex`](../packages/hooks/hooks-codex) | Maintenance mode: hybrid generated: Cordis event declarations and most producer/listener edges are AST-scanned; dynamic dispatch sites are classified in `scripts/gen-doc-graphs.ts`. diff --git a/docs/module-graph.md b/docs/module-graph.md index ed1cc592d4..8283a8b756 100644 --- a/docs/module-graph.md +++ b/docs/module-graph.md @@ -171,6 +171,8 @@ flowchart TD pkg_subagent_inprocess --> pkg_llm pkg_subagent_inprocess --> pkg_session pkg_subagent_inprocess --> pkg_subagent + pkg_subagent_inprocess --> pkg_system_prompt + pkg_subagent_inprocess --> pkg_tools pkg_tool_subagent --> pkg_agent pkg_tool_subagent --> pkg_llm pkg_tool_subagent --> pkg_subagent @@ -241,7 +243,7 @@ flowchart TD | [`acp`](../packages/ui/acp) | `ui` | [`agent`](../packages/core/agent), [`llm`](../packages/llm/llm), [`session`](../packages/core/session), [`session-persistence`](../packages/session-persistence/session-persistence), [`tools`](../packages/core/tools) | | [`agent-core`](../packages/core/agent-core) | `core` | [`agent`](../packages/core/agent), [`agent-loop`](../packages/core/agent-loop), [`invariants`](../packages/support/invariants), [`llm`](../packages/llm/llm), [`session`](../packages/core/session), [`system-prompt`](../packages/core/system-prompt), [`tool-bash`](../packages/bash/tool-bash), [`tools`](../packages/core/tools) | | [`subagent-acp`](../packages/subagent/subagent-acp) | `subagent` | [`agent`](../packages/core/agent), [`llm`](../packages/llm/llm), [`subagent`](../packages/subagent/subagent) | -| [`subagent-inprocess`](../packages/subagent/subagent-inprocess) | `subagent` | [`agent`](../packages/core/agent), [`llm`](../packages/llm/llm), [`session`](../packages/core/session), [`subagent`](../packages/subagent/subagent) | +| [`subagent-inprocess`](../packages/subagent/subagent-inprocess) | `subagent` | [`agent`](../packages/core/agent), [`llm`](../packages/llm/llm), [`session`](../packages/core/session), [`subagent`](../packages/subagent/subagent), [`system-prompt`](../packages/core/system-prompt), [`tools`](../packages/core/tools) | | [`tool-subagent`](../packages/subagent/tool-subagent) | `subagent` | [`agent`](../packages/core/agent), [`llm`](../packages/llm/llm), [`subagent`](../packages/subagent/subagent), [`tools`](../packages/core/tools) | | [`hooks-claude`](../packages/hooks/hooks-claude) | `hooks` | [`agent`](../packages/core/agent), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm`](../packages/llm/llm), [`session`](../packages/core/session), [`subagent`](../packages/subagent/subagent), [`tools`](../packages/core/tools) | | [`subagent-mock`](../packages/support/subagent-mock) | `support` | [`agent`](../packages/core/agent), [`llm`](../packages/llm/llm), [`subagent`](../packages/subagent/subagent) | diff --git a/packages/core/tools/README.md b/packages/core/tools/README.md index 65039aea58..d43f24e123 100644 --- a/packages/core/tools/README.md +++ b/packages/core/tools/README.md @@ -71,6 +71,12 @@ A `defineTool` tool also **validates the model-generated arguments against its ` See `defineTool`, `validateArgs`, `ToolArgsError`, `SchemaSpec`, `InferArgs`, and `schemaSpecToJsonSchema` in the public API for details. +### Structured-output schema subset + +A separate vocabulary for callers that DEMAND a machine-readable value from an agent — the subagent seam's `SubagentStartRequest.outputSchema` (and, by extension, a workflow's `agent({ schema })`). Unlike `SchemaSpec` (the author-facing DSL for tool parameters), a `StructuredOutputSchema` is an object-rooted **raw JSON Schema subset** as data: it travels verbatim to the model as a forced tool's `parameters`, and the produced value is validated against it. + +The subset is deliberately narrow and REJECTS LOUD outside it — accepting a keyword the validator doesn't enforce would validate less than the schema promises (accepted-then-ignored). Supported: single-string `type` (`object`/`array`/`string`/`number`/`integer`/`boolean`/`null`; type arrays rejected), `properties`/`required`/`additionalProperties` (boolean; every `required` key must be declared), `items`, scalar-only `enum`/`const`; annotations (`description`/`title`/`default`/`examples`) are ignored but must still be JSON data. `assertSupportedOutputSchema(schema)` throws `OutputSchemaError` (`code: 'UNSUPPORTED_SCHEMA'`, listing every violation) for anything else; `validateStructuredValue(schema, value)` returns path-qualified violations (empty = valid, total — never throws). + ### Tool-owned UI presentation A tool owns how ITS calls render in a UI (an editor's tool-call card, a CLI log line) — a UI plugin must NOT special-case tool names. A `ToolDefinition` may declare two optional, pure, display-only methods that return a **`card`-tagged render intent** (a discriminated union — a tool declares its card kind once and a UI bridge switches on `card`): diff --git a/packages/core/tools/src/index.ts b/packages/core/tools/src/index.ts index dd0ed918db..39dafd6f1a 100644 --- a/packages/core/tools/src/index.ts +++ b/packages/core/tools/src/index.ts @@ -28,6 +28,16 @@ export { type JsonSchemaObject, } from './schema.ts' +export { + assertSupportedOutputSchema, + validateStructuredValue, + OutputSchemaError, + type StructuredOutputSchema, + type StructuredSchemaNode, + type StructuredSchemaType, + type StructuredScalar, +} from './json-schema.ts' + // The render-intent vocabulary a tool declares via `presentCall`/`presentResult` // lives in its own UI-facing module; re-export it so `@deepseek-ai/dsh-tools` // stays the single public surface for consumers (producers + the ACP bridge). diff --git a/packages/core/tools/src/json-schema.ts b/packages/core/tools/src/json-schema.ts new file mode 100644 index 0000000000..35eeb240a4 --- /dev/null +++ b/packages/core/tools/src/json-schema.ts @@ -0,0 +1,322 @@ +/** + * Structured-output JSON Schema subset: the vocabulary a caller uses to demand + * a machine-readable result from a subagent (`SubagentStartRequest.outputSchema`) + * or a workflow `agent()` call. + * + * This is deliberately NOT full JSON Schema. The schema travels verbatim to the + * model as a forced tool's `parameters`, and the value the model produces is + * validated here — so every accepted keyword must be one this module actually + * enforces. Accepting a keyword we don't enforce would validate less than the + * schema promises (accepted-then-ignored), so anything outside the subset is + * REJECTED LOUD by {@link assertSupportedOutputSchema} instead. The subset: + * + * - `type` — a single string (`object`/`array`/`string`/`number`/`integer`/ + * `boolean`/`null`); type ARRAYS (`["string","null"]`) are rejected. + * - `properties`/`required`/`additionalProperties` (boolean) on objects; every + * `required` key must be declared in `properties`. `additionalProperties` + * absent keeps standard JSON Schema semantics (extra keys allowed). + * - `items` on arrays (absent ⇒ any JSON items). + * - `enum` (non-empty, scalars only) and `const` (scalar) on scalar types. + * - Annotations `description`/`title`/`default`/`examples` are allowed and + * ignored (they constrain nothing), except that they must still be JSON data + * — the schema is serialized onto the wire, so a non-JSON annotation would be + * silently mangled. + * + * Values checked by {@link validateStructuredValue} are expected to be plain + * host-realm JSON data (model tool-call arguments are parsed wire JSON; a + * caller holding foreign-realm data materializes it first). + * + * @module dsh-tools/json-schema + */ + +import { assertNever, HarnessError } from '@deepseek-ai/dsh-llm' + +/** The scalar values `enum`/`const` may carry (finite numbers only). */ +export type StructuredScalar = string | number | boolean | null + +/** The `type` keywords the subset accepts. */ +export type StructuredSchemaType = 'object' | 'array' | 'string' | 'number' | 'integer' | 'boolean' | 'null' + +/** + * One node of the structured-output schema subset. Recursive via `properties` + * and `items`; see the module doc for the exact keyword semantics. + */ +export interface StructuredSchemaNode { + type: StructuredSchemaType + /** Nested property schemas (`type: 'object'` only). */ + properties?: Record + /** Required property names; each must appear in `properties`. */ + required?: string[] + /** `false` rejects undeclared keys; absent/`true` allows them (JSON Schema default). */ + additionalProperties?: boolean + /** Item schema (`type: 'array'` only); absent ⇒ any JSON items. */ + items?: StructuredSchemaNode + /** Allowed values (scalar types only). */ + enum?: StructuredScalar[] + /** The single allowed value (scalar types only). */ + const?: StructuredScalar + /** Annotation, ignored for validation. */ + description?: string + /** Annotation, ignored for validation. */ + title?: string + /** Annotation, ignored for validation (must still be JSON data). */ + default?: unknown + /** Annotation, ignored for validation (must still be JSON data). */ + examples?: unknown +} + +/** A structured-output schema: an OBJECT-rooted {@link StructuredSchemaNode}. */ +export type StructuredOutputSchema = StructuredSchemaNode & { type: 'object' } + +/** + * Thrown by {@link assertSupportedOutputSchema} when a schema falls outside the + * supported subset. Extends {@link HarnessError} (`code: 'UNSUPPORTED_SCHEMA'`) + * so seam code and tool results can route on it; `violations` lists every + * offending path, not just the first. + */ +export class OutputSchemaError extends HarnessError { + /** The individual violation messages, in walk order. */ + readonly violations: string[] + + constructor(violations: string[]) { + super(`unsupported output schema: ${violations.join('; ')}`, 'UNSUPPORTED_SCHEMA') + this.name = 'OutputSchemaError' + this.violations = violations + } +} + +/** The keywords the subset accepts, checked (`constraint`) or ignored (`annotation`). */ +const CONSTRAINT_KEYWORDS = new Set(['type', 'properties', 'required', 'additionalProperties', 'items', 'enum', 'const']) +const ANNOTATION_KEYWORDS = new Set(['description', 'title', 'default', 'examples']) + +const SCHEMA_TYPES: readonly StructuredSchemaType[] = ['object', 'array', 'string', 'number', 'integer', 'boolean', 'null'] + +/** Whether a value is a non-null, non-array object (structural, realm-agnostic). */ +function isObjectLike(value: unknown): value is Record { + return typeof value === 'object' && value !== null && !Array.isArray(value) +} + +/** Whether a value is a supported scalar (`enum`/`const` member): string, finite number, boolean, or null. */ +function isStructuredScalar(value: unknown): value is StructuredScalar { + return value === null || typeof value === 'string' || typeof value === 'boolean' + || (typeof value === 'number' && Number.isFinite(value)) +} + +/** + * Whether a value is JSON data (annotation payloads only): scalars, arrays, and + * object-likes of such values. Realm-agnostic on purpose (no prototype check) — + * the schema may have been materialized from another realm; structural JSON-ness + * is what the wire needs. Cycles are rejected via `seen`. + */ +function isJsonData(value: unknown, seen: Set): boolean { + if (isStructuredScalar(value)) return true + // The scalar check above already returned for null, so `object` here is a real object. + if (typeof value !== 'object') return false + if (seen.has(value)) return false + seen.add(value) + try { + if (Array.isArray(value)) return value.every(entry => isJsonData(entry, seen)) + return Object.values(value).every(entry => isJsonData(entry, seen)) + } finally { + seen.delete(value) + } +} + +/** Collect subset violations for one schema node (recursive walk). */ +function checkSchemaNode(node: unknown, path: string, violations: string[], seen: Set): void { + if (!isObjectLike(node)) { + violations.push(`${path} must be a schema object`) + return + } + if (seen.has(node)) { + violations.push(`${path} is circular`) + return + } + seen.add(node) + + for (const key of Object.keys(node)) { + if (CONSTRAINT_KEYWORDS.has(key)) continue + if (ANNOTATION_KEYWORDS.has(key)) { + if (!isJsonData(node[key], new Set())) violations.push(`${path}.${key} annotation must be JSON data`) + continue + } + violations.push(`${path}.${key} is not a supported keyword (subset: type/properties/required/additionalProperties/items/enum/const + annotations)`) + } + if (typeof node.description !== 'undefined' && typeof node.description !== 'string') { + violations.push(`${path}.description must be a string`) + } + if (typeof node.title !== 'undefined' && typeof node.title !== 'string') { + violations.push(`${path}.title must be a string`) + } + + const type = node.type + if (typeof type !== 'string' || !(SCHEMA_TYPES as readonly unknown[]).includes(type)) { + violations.push(Array.isArray(type) + ? `${path}.type must be a single type string (type arrays are not supported)` + : `${path}.type must be one of ${SCHEMA_TYPES.join('/')}`) + seen.delete(node) + return + } + const schemaType = type as StructuredSchemaType + + // Keywords that only make sense on one type are rejected elsewhere — an + // `items` on an object (or `properties` on a string) is a schema-author bug + // the subset surfaces rather than ignores. + const allowedFor: Record = { + properties: ['object'], + required: ['object'], + additionalProperties: ['object'], + items: ['array'], + enum: ['string', 'number', 'integer', 'boolean', 'null'], + const: ['string', 'number', 'integer', 'boolean', 'null'], + } + for (const [key, types] of Object.entries(allowedFor)) { + if (key in node && !types.includes(schemaType)) { + violations.push(`${path}.${key} is not supported on type "${schemaType}"`) + } + } + + switch (schemaType) { + case 'object': { + const properties = node.properties + if (properties !== undefined) { + if (!isObjectLike(properties)) { + violations.push(`${path}.properties must be an object of schemas`) + } else { + for (const [key, child] of Object.entries(properties)) { + checkSchemaNode(child, `${path}.properties.${key}`, violations, seen) + } + } + } + const required = node.required + if (required !== undefined) { + if (!Array.isArray(required) || required.some(entry => typeof entry !== 'string')) { + violations.push(`${path}.required must be an array of strings`) + } else { + const declared = isObjectLike(properties) ? properties : {} + for (const key of required) { + if (!(key in declared)) violations.push(`${path}.required names "${key}" which is not in properties`) + } + } + } + if (node.additionalProperties !== undefined && typeof node.additionalProperties !== 'boolean') { + violations.push(`${path}.additionalProperties must be a boolean`) + } + break + } + case 'array': { + if (node.items !== undefined) checkSchemaNode(node.items, `${path}.items`, violations, seen) + break + } + case 'string': + case 'number': + case 'integer': + case 'boolean': + case 'null': { + const allowed = node.enum + if (allowed !== undefined) { + if (!Array.isArray(allowed) || allowed.length === 0 || !allowed.every(entry => isStructuredScalar(entry))) { + violations.push(`${path}.enum must be a non-empty array of scalars`) + } + } + if ('const' in node && !isStructuredScalar(node.const)) { + violations.push(`${path}.const must be a scalar`) + } + break + } + /* v8 ignore start -- defensive: schemaType was membership-checked against SCHEMA_TYPES above, so no runtime value reaches here */ + default: + assertNever(schemaType, 'assertSupportedOutputSchema') + /* v8 ignore stop */ + } + + seen.delete(node) +} + +/** + * Assert `schema` is a supported {@link StructuredOutputSchema} — object-rooted + * and entirely within the enforced subset. Throws {@link OutputSchemaError} + * (`UNSUPPORTED_SCHEMA`) listing EVERY violation; returns (and narrows) on + * success. Call this at the seam boundary, before any child is created. + * @param schema - the caller-supplied schema (unknown until asserted). + */ +export function assertSupportedOutputSchema(schema: unknown): asserts schema is StructuredOutputSchema { + const violations: string[] = [] + checkSchemaNode(schema, 'schema', violations, new Set()) + if (violations.length === 0 && (schema as StructuredSchemaNode).type !== 'object') { + violations.push('schema.type must be "object" (structured output is object-rooted)') + } + if (violations.length > 0) throw new OutputSchemaError(violations) +} + +/** Collect violations for one value against an (already asserted) schema node. */ +function checkValue(node: StructuredSchemaNode, value: unknown, path: string): string[] { + switch (node.type) { + case 'object': { + if (!isObjectLike(value)) return [`"${path}" must be an object`] + const violations: string[] = [] + const properties = node.properties ?? {} + for (const key of node.required ?? []) { + if (value[key] === undefined) violations.push(`missing required property "${path}.${key}"`) + } + for (const [key, child] of Object.entries(properties)) { + if (value[key] === undefined) continue + violations.push(...checkValue(child, value[key], `${path}.${key}`)) + } + if (node.additionalProperties === false) { + for (const key of Object.keys(value)) { + if (!(key in properties)) violations.push(`"${path}.${key}" is not a declared property (additionalProperties: false)`) + } + } + return violations + } + case 'array': { + if (!Array.isArray(value)) return [`"${path}" must be an array`] + if (!node.items) return [] + const items = node.items + return value.flatMap((entry, index) => checkValue(items, entry, `${path}[${index}]`)) + } + case 'string': { + if (typeof value !== 'string') return [`"${path}" must be a string`] + break + } + case 'number': { + if (typeof value !== 'number' || !Number.isFinite(value)) return [`"${path}" must be a finite number`] + break + } + case 'integer': { + if (typeof value !== 'number' || !Number.isInteger(value)) return [`"${path}" must be an integer`] + break + } + case 'boolean': { + if (typeof value !== 'boolean') return [`"${path}" must be a boolean`] + break + } + case 'null': { + if (value !== null) return [`"${path}" must be null`] + break + } + default: + return assertNever(node.type, 'validateStructuredValue') + } + // Scalar constraint checks, shared by every scalar branch above. + if (node.enum && !node.enum.includes(value)) { + return [`"${path}" must be one of ${JSON.stringify(node.enum)}`] + } + if ('const' in node && value !== node.const) { + return [`"${path}" must be ${JSON.stringify(node.const)}`] + } + return [] +} + +/** + * Validate a value against an (already {@link assertSupportedOutputSchema}- + * asserted) schema. Returns human-readable, path-qualified violation messages + * — empty means valid. Total: never throws, however malformed the value. + * @param schema - the asserted schema to check against. + * @param value - the candidate value (e.g. parsed tool-call arguments). + * @returns every violation found, in walk order (empty = valid). + */ +export function validateStructuredValue(schema: StructuredOutputSchema, value: unknown): string[] { + return checkValue(schema, value, 'value') +} diff --git a/packages/core/tools/tests/json-schema.spec.ts b/packages/core/tools/tests/json-schema.spec.ts new file mode 100644 index 0000000000..e7635b06f3 --- /dev/null +++ b/packages/core/tools/tests/json-schema.spec.ts @@ -0,0 +1,254 @@ +import { describe, expect, it } from 'vitest' +import { + assertSupportedOutputSchema, + OutputSchemaError, + validateStructuredValue, + type StructuredOutputSchema, +} from '../src/json-schema.ts' + +/** Assert-and-narrow helper: the asserted schema, typed. */ +function asserted(schema: unknown): StructuredOutputSchema { + assertSupportedOutputSchema(schema) + return schema +} + +/** The violations OutputSchemaError carries for a bad schema (throws if it passes). */ +function violationsOf(schema: unknown): string[] { + try { + assertSupportedOutputSchema(schema) + } catch (error: unknown) { + if (error instanceof OutputSchemaError) return error.violations + throw error + } + throw new Error('expected the schema to be rejected') +} + +describe('assertSupportedOutputSchema', () => { + it('accepts a representative subset schema (all supported keywords)', () => { + const schema = asserted({ + type: 'object', + description: 'a finding', + title: 'Finding', + properties: { + file: { type: 'string', description: 'path' }, + line: { type: 'integer' }, + severity: { type: 'string', enum: ['low', 'high'] }, + kind: { type: 'string', const: 'bug' }, + score: { type: 'number' }, + confirmed: { type: 'boolean' }, + parent: { type: 'null' }, + tags: { type: 'array', items: { type: 'string' } }, + nested: { + type: 'object', + properties: { x: { type: 'number', default: 3, examples: [1, 2] } }, + additionalProperties: false, + }, + anything: { type: 'array' }, + }, + required: ['file', 'line'], + additionalProperties: true, + }) + expect(schema.type).toBe('object') + }) + + it('rejects a non-object root (scalar/array-rooted schemas)', () => { + expect(violationsOf({ type: 'string' })).toEqual(['schema.type must be "object" (structured output is object-rooted)']) + expect(violationsOf({ type: 'array', items: { type: 'string' } })) + .toContain('schema.type must be "object" (structured output is object-rooted)') + }) + + it('rejects non-object schema nodes and missing/unknown type', () => { + expect(violationsOf('nope')).toEqual(['schema must be a schema object']) + expect(violationsOf(null)).toEqual(['schema must be a schema object']) + expect(violationsOf([])).toEqual(['schema must be a schema object']) + expect(violationsOf({})).toEqual(['schema.type must be one of object/array/string/number/integer/boolean/null']) + expect(violationsOf({ type: 'tuple' })[0]).toMatch(/type must be one of/) + expect(violationsOf({ type: 'object', properties: { a: 'str' } })).toEqual(['schema.properties.a must be a schema object']) + }) + + it('rejects type ARRAYS with a dedicated message', () => { + expect(violationsOf({ type: ['string', 'null'] })) + .toEqual(['schema.type must be a single type string (type arrays are not supported)']) + }) + + it('rejects unsupported constraint keywords loudly (never accepted-then-ignored)', () => { + for (const keyword of ['oneOf', 'anyOf', 'allOf', 'not', 'pattern', 'minimum', 'maxLength', '$ref']) { + const bad = violationsOf({ type: 'object', [keyword]: [] }) + expect(bad.some(v => v.includes(`schema.${keyword} is not a supported keyword`))).toBe(true) + } + }) + + it('reports EVERY violation, not just the first', () => { + const bad = violationsOf({ + type: 'object', + pattern: 'x', + properties: { a: { type: 'weird' }, b: { type: 'string', minimum: 1 } }, + }) + expect(bad.length).toBe(3) + }) + + it('rejects keywords on the wrong type (items on object, properties on string, enum on object)', () => { + expect(violationsOf({ type: 'object', items: { type: 'string' } })) + .toEqual(['schema.items is not supported on type "object"']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'string', properties: {} } } })) + .toEqual(['schema.properties.a.properties is not supported on type "string"']) + expect(violationsOf({ type: 'object', enum: [1] })) + .toEqual(['schema.enum is not supported on type "object"']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'array', const: 1 } } })) + .toEqual(['schema.properties.a.const is not supported on type "array"']) + }) + + it('validates required: must be string[] naming declared properties', () => { + expect(violationsOf({ type: 'object', required: 'file' })) + .toEqual(['schema.required must be an array of strings']) + expect(violationsOf({ type: 'object', required: [1] })) + .toEqual(['schema.required must be an array of strings']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'string' } }, required: ['b'] })) + .toEqual(['schema.required names "b" which is not in properties']) + expect(violationsOf({ type: 'object', required: ['a'] })) + .toEqual(['schema.required names "a" which is not in properties']) + }) + + it('validates additionalProperties must be boolean and enum/const must be scalars', () => { + expect(violationsOf({ type: 'object', additionalProperties: {} })) + .toEqual(['schema.additionalProperties must be a boolean']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'string', enum: [] } } })) + .toEqual(['schema.properties.a.enum must be a non-empty array of scalars']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'string', enum: [{}] } } })) + .toEqual(['schema.properties.a.enum must be a non-empty array of scalars']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'string', enum: 'x' } } })) + .toEqual(['schema.properties.a.enum must be a non-empty array of scalars']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'number', enum: [Number.NaN] } } })) + .toEqual(['schema.properties.a.enum must be a non-empty array of scalars']) + expect(violationsOf({ type: 'object', properties: { a: { type: 'string', const: {} } } })) + .toEqual(['schema.properties.a.const must be a scalar']) + }) + + it('rejects non-string description/title and non-JSON annotation payloads', () => { + expect(violationsOf({ type: 'object', description: 7 })) + .toEqual(['schema.description must be a string']) + expect(violationsOf({ type: 'object', title: 7 })) + .toEqual(['schema.title must be a string']) + expect(violationsOf({ type: 'object', default: () => 1 })) + .toEqual(['schema.default annotation must be JSON data']) + expect(violationsOf({ type: 'object', examples: [undefined] })) + .toEqual(['schema.examples annotation must be JSON data']) + expect(violationsOf({ type: 'object', examples: [Number.POSITIVE_INFINITY] })) + .toEqual(['schema.examples annotation must be JSON data']) + // A cyclic annotation payload is caught by the JSON-data walk. + const cyclicAnnotation: Record = {} + cyclicAnnotation.self = cyclicAnnotation + expect(violationsOf({ type: 'object', default: cyclicAnnotation })) + .toEqual(['schema.default annotation must be JSON data']) + // Object/array annotations that ARE JSON data pass. + asserted({ type: 'object', default: { a: [1, 'x', null, true] } }) + }) + + it('rejects a circular schema instead of recursing forever', () => { + const node: Record = { type: 'object' } + node.properties = { self: node } + expect(violationsOf(node)).toEqual(['schema.properties.self is circular']) + }) + + it('accepts the same subschema object reused in two SIBLING positions (a DAG, not a cycle)', () => { + const leaf = { type: 'string' } + asserted({ type: 'object', properties: { a: leaf, b: leaf } }) + }) +}) + +describe('validateStructuredValue', () => { + const schema = asserted({ + type: 'object', + properties: { + file: { type: 'string' }, + line: { type: 'integer' }, + score: { type: 'number' }, + confirmed: { type: 'boolean' }, + parent: { type: 'null' }, + severity: { type: 'string', enum: ['low', 'high'] }, + kind: { type: 'string', const: 'bug' }, + tags: { type: 'array', items: { type: 'string' } }, + free: { type: 'array' }, + nested: { type: 'object', properties: { x: { type: 'number' } }, required: ['x'], additionalProperties: false }, + }, + required: ['file'], + }) + + it('accepts a fully valid value (empty violations)', () => { + expect(validateStructuredValue(schema, { + file: 'a.ts', line: 3, score: 0.5, confirmed: true, parent: null, + severity: 'high', kind: 'bug', tags: ['x'], free: [1, { any: true }], nested: { x: 1 }, + })).toEqual([]) + }) + + it('reports missing required and wrong root type', () => { + expect(validateStructuredValue(schema, {})).toEqual(['missing required property "value.file"']) + expect(validateStructuredValue(schema, 'nope')).toEqual(['"value" must be an object']) + expect(validateStructuredValue(schema, [])).toEqual(['"value" must be an object']) + }) + + it('type-checks every scalar branch with path-qualified messages', () => { + expect(validateStructuredValue(schema, { file: 1 })).toEqual(['"value.file" must be a string']) + expect(validateStructuredValue(schema, { file: 'a', line: 1.5 })).toEqual(['"value.line" must be an integer']) + expect(validateStructuredValue(schema, { file: 'a', line: 'x' })).toEqual(['"value.line" must be an integer']) + expect(validateStructuredValue(schema, { file: 'a', score: 'x' })).toEqual(['"value.score" must be a finite number']) + expect(validateStructuredValue(schema, { file: 'a', score: Number.NaN })).toEqual(['"value.score" must be a finite number']) + expect(validateStructuredValue(schema, { file: 'a', confirmed: 'yes' })).toEqual(['"value.confirmed" must be a boolean']) + expect(validateStructuredValue(schema, { file: 'a', parent: 0 })).toEqual(['"value.parent" must be null']) + }) + + it('enforces enum membership and const equality', () => { + expect(validateStructuredValue(schema, { file: 'a', severity: 'mid' })) + .toEqual(['"value.severity" must be one of ["low","high"]']) + expect(validateStructuredValue(schema, { file: 'a', kind: 'feature' })) + .toEqual(['"value.kind" must be "bug"']) + }) + + it('checks arrays per index; an items-less array accepts anything', () => { + expect(validateStructuredValue(schema, { file: 'a', tags: 'x' })).toEqual(['"value.tags" must be an array']) + expect(validateStructuredValue(schema, { file: 'a', tags: ['ok', 2] })).toEqual(['"value.tags[1]" must be a string']) + expect(validateStructuredValue(schema, { file: 'a', free: [{ deep: [1] }, null] })).toEqual([]) + }) + + it('recurses into nested objects: required + additionalProperties: false', () => { + expect(validateStructuredValue(schema, { file: 'a', nested: {} })) + .toEqual(['missing required property "value.nested.x"']) + expect(validateStructuredValue(schema, { file: 'a', nested: { x: 1, y: 2 } })) + .toEqual(['"value.nested.y" is not a declared property (additionalProperties: false)']) + expect(validateStructuredValue(schema, { file: 'a', nested: 3 })) + .toEqual(['"value.nested" must be an object']) + }) + + it('a required key present-but-undefined counts as missing', () => { + expect(validateStructuredValue(schema, { file: undefined })).toEqual(['missing required property "value.file"']) + }) + + it('collects multiple violations across branches in one pass', () => { + expect(validateStructuredValue(schema, { line: 'x', severity: 'mid' })).toEqual([ + 'missing required property "value.file"', + '"value.line" must be an integer', + '"value.severity" must be one of ["low","high"]', + ]) + }) + + it('null-typed const/enum work through the scalar path', () => { + const nullish = asserted({ type: 'object', properties: { a: { type: 'null', const: null } } }) + expect(validateStructuredValue(nullish, { a: null })).toEqual([]) + }) + + it('rejects a non-object properties value in the schema walk', () => { + expect(violationsOf({ type: 'object', properties: [] })) + .toEqual(['schema.properties must be an object of schemas']) + }) + + it('an object schema without properties/required only type-checks its value', () => { + const bare = asserted({ type: 'object' }) + expect(validateStructuredValue(bare, { any: ['thing'] })).toEqual([]) + expect(validateStructuredValue(bare, 7)).toEqual(['"value" must be an object']) + }) + + it('validateStructuredValue throws on a type the assert would never let through (assertNever backstop)', () => { + const forged = { type: 'tuple' } as unknown as StructuredOutputSchema + expect(() => validateStructuredValue(forged, 1)).toThrow(/tuple/) + }) +}) diff --git a/packages/subagent/subagent-fork/README.md b/packages/subagent/subagent-fork/README.md index c691d56355..7b43f82261 100644 --- a/packages/subagent/subagent-fork/README.md +++ b/packages/subagent/subagent-fork/README.md @@ -12,12 +12,13 @@ The seam this rides on: `CreateAgentOptions.seed` (added on `dsh-agent`, threade ## Capabilities -`{ outputSchema: false, depthLimit: true, toolFilter: false }` — identical to spawn (the depth/model/output behavior is the shared driver's). +`{ outputSchema: true, depthLimit: true, toolFilter: false }` — identical to spawn (the depth/model/structured-output behavior is the shared driver's). ## Config | Key | Meaning | |---|---| | `providerName` | Registry name on `ctx.subagents` (default `fork`). | +| `structuredNudgeRetries` | How many times a structured run re-prompts a child that finished cleanly without calling `structured_output` (default 1). | See [`dsh-subagent-spawn`](../subagent-spawn/README.md) for the run lifecycle, model inheritance, and depth tracking — all shared. diff --git a/packages/subagent/subagent-fork/src/index.ts b/packages/subagent/subagent-fork/src/index.ts index b6d0c10e44..10d0492198 100644 --- a/packages/subagent/subagent-fork/src/index.ts +++ b/packages/subagent/subagent-fork/src/index.ts @@ -25,19 +25,29 @@ import z from 'schemastery' import type { SessionEvent } from '@deepseek-ai/dsh-session' import type { Agent } from '@deepseek-ai/dsh-agent' import type { SubagentCapabilities, SubagentProvider, SubagentStartRequest } from '@deepseek-ai/dsh-subagent' -import { startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' +import { acquireStructuredRuntime, startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' export const name = 'subagent-fork' +// `tools` is deliberately NOT injected — same rationale as subagent-spawn: the +// structured runtime gates its capture-tool registration on `tools` itself, so +// this backend's apply timing (and the delegation tool's position in the +// model-visible tool list) is unchanged by structured output. export const inject = ['subagents', 'agents'] -/** Config: the registry name to register the provider under. */ +/** Config: the registry name to register the provider under, plus structured-run tuning. */ export interface Config { /** Provider name on `ctx.subagents` (default `fork`). */ providerName: string + /** + * How many times a structured run re-prompts a child that finished cleanly + * without calling `structured_output` before giving up (default 1). + */ + structuredNudgeRetries: number } export const Config: z = z.object({ providerName: z.string().default('fork'), + structuredNudgeRetries: z.natural().default(1), }) /** @@ -57,20 +67,26 @@ export function completedTurnPrefix(parent: Agent): SessionEvent[] { } /** - * The fork provider. Supports `depthLimit`; NOT `outputSchema`/`toolFilter` this - * cut (the service rejects a request needing either before `start` runs). + * The fork provider. Supports `depthLimit` and `outputSchema` (via the shared + * in-process structured runtime); NOT `toolFilter` this cut (the service + * rejects a request needing it before `start` runs). */ class ForkProvider implements SubagentProvider { - readonly capabilities: SubagentCapabilities = { outputSchema: false, depthLimit: true, toolFilter: false } + readonly capabilities: SubagentCapabilities = { outputSchema: true, depthLimit: true, toolFilter: false } // Context contract: a forked child IS seeded with the parent's completed-turn prefix. readonly inheritsParentContext = true - constructor(readonly name: string, private readonly ctx: Context) {} + constructor( + readonly name: string, + private readonly ctx: Context, + private readonly structuredNudgeRetries: number, + ) {} start(request: SubagentStartRequest) { const seed = completedTurnPrefix(request.parent) return startInProcessRun(this.ctx, request, { providerName: this.name, + structuredNudgeRetries: this.structuredNudgeRetries, // Only pass a seed when there's a completed turn to inherit; an empty seed // is equivalent to a fresh child, so omit it to keep the session unseeded. ...seed.length > 0 ? { seed } : {}, @@ -79,5 +95,12 @@ class ForkProvider implements SubagentProvider { } export function apply(ctx: Context, config: Config): void { - ctx.subagents.registerProvider(new ForkProvider(config.providerName, ctx)) + // Hold the structured runtime for the plugin's lifetime (see the spawn + // backend — same two-level lifetime: backends for availability, runs for + // mid-run survival across a backend unload). + ctx.effect(() => { + const acquisition = acquireStructuredRuntime(ctx) + return () => { acquisition.release() } + }, 'subagent-fork structured runtime') + ctx.subagents.registerProvider(new ForkProvider(config.providerName, ctx, config.structuredNudgeRetries)) } diff --git a/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts b/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts index 1f932fbaf9..82caf25948 100644 --- a/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts +++ b/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts @@ -30,8 +30,8 @@ async function setup(script: Script) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - await ctx.plugin(Spawn, { providerName: 'spawn' }) - await ctx.plugin(fork, { providerName: 'fork' }) + await ctx.plugin(Spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) + await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: 1 }) ctx.llm.registerAdapter(['mock'], new MockAdapter(script)) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) return { ctx, parent } diff --git a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts index 56441a656f..2545188b52 100644 --- a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts +++ b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts @@ -37,7 +37,7 @@ async function setup(script: Script) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - await ctx.plugin(fork, { providerName: 'fork' }) + await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: 1 }) ctx.llm.registerAdapter(['mock'], new MockAdapter(script)) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) return { ctx, parent } @@ -161,16 +161,22 @@ describe('dsh-subagent-fork', () => { await run.dispose() }) - it('advertises depthLimit but not outputSchema/toolFilter', async () => { + it('advertises depthLimit and outputSchema but not toolFilter', async () => { const { ctx } = await setup([]) - expect(ctx.subagents.getProvider('fork')!.capabilities).toEqual({ outputSchema: false, depthLimit: true, toolFilter: false }) + expect(ctx.subagents.getProvider('fork')!.capabilities).toEqual({ outputSchema: true, depthLimit: true, toolFilter: false }) }) it('unregisters the provider when its fiber is disposed (HMR safety)', async () => { const ctx = new Context() await ctx.plugin(SubagentService) await ctx.plugin(AgentRegistry) - const fiber = await ctx.plugin(fork, { providerName: 'fork' }) + // The backend does NOT inject 'tools' (the structured runtime gates its + // capture-tool registration on tools availability itself, keeping backend + // apply timing — and the delegation tool's prompt position — unchanged); + // the registries are loaded here so the runtime registers eagerly anyway. + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + const fiber = await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: 1 }) expect(ctx.subagents.list()).toEqual(['fork']) await fiber.dispose() expect(ctx.subagents.list()).toEqual([]) diff --git a/packages/subagent/subagent-inprocess/README.md b/packages/subagent/subagent-inprocess/README.md index af6d5792a9..b816da8946 100644 --- a/packages/subagent/subagent-inprocess/README.md +++ b/packages/subagent/subagent-inprocess/README.md @@ -8,16 +8,27 @@ The shared **in-process subagent run driver**. A pure library (no provider, no r Runs a child as a child [`Agent`](../../core/agent) on the same cordis context (`ctx.agents`): -1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability); -2. creates a child via `ctx.agents.create` with a fresh `AgentId`/`SessionId`, the parent's `cwd` + `parentSession` lineage, the optional `options.seed` (fork's completed-turn prefix; omitted for a fresh child), and `agentOptions` (the child inherits the **parent's model** by default — a child with no model can't run — overridable via `request.agentOptions.model`; the system prompt is NOT inherited); -3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts); -4. reads the result, scoped to the child's OWN events (everything at or after `seedLength`, so a seeded child that produced no message of its own never returns the seeded parent's last message): the last `assistant/message` content (deep-cloned — the log is frozen) and the last `turn/end.reason` mapped to a `SubagentStopReason`. +1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability); a `request.outputSchema` is asserted against the supported subset (`assertSupportedOutputSchema` from [dsh-tools](../../core/tools/README.md)) before any child exists; +2. creates a child via `ctx.agents.create` with a fresh `AgentId`/`SessionId`, the parent's `cwd` + `parentSession` lineage, the optional `options.seed` (fork's completed-turn prefix; omitted for a fresh child), and `agentOptions` (the child inherits the **parent's model** by default — a child with no model can't run — overridable via `request.agentOptions.model`; the deployment persona needs no inheritance — it is a context-wide prompt section); +3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts); a structured child that finished a turn CLEANLY without calling `structured_output` is re-prompted (a nudge — a fresh turn) up to `options.structuredNudgeRetries` times; +4. reads the result, scoped to the child's OWN events (everything at or after `seedLength`, so a seeded child that produced no message of its own never returns the seeded parent's last message): the last `assistant/message` content (deep-cloned — the log is frozen) and the last `turn/end.reason` mapped to a `SubagentStopReason`. A structured run surfaces the captured value as `result.structured`; a structured child that finished cleanly WITHOUT ever capturing settles `error` (a clean finish without the demanded result is a failure, not a success with a missing field). `dispose()` delegates to `AgentHandle.dispose()` (stop loop → await quiescence → remove session); `cancel()` cancels the child's in-flight turn. A cancel landing before any `turn/end` (the pre-turn window) still settles `aborted`, honoring the cancel contract rather than the generic no-turn `error`. ### `InProcessRunOptions` -`{ providerName: string; seed?: SessionEvent[] }` — the per-backend inputs: the provider name (for error context) and the optional child-session seed. +`{ providerName: string; seed?: SessionEvent[]; structuredNudgeRetries: number }` — the per-backend inputs: the provider name (for error context), the optional child-session seed, and the structured-run nudge budget (REQUIRED, resolved from the backend's validated Config — the driver never fills it with a hidden default). + +### Structured output: `acquireStructuredRuntime(ctx): StructuredAcquisition` + +The mechanism behind `outputSchema` for in-process children. One globally registered `structured_output` capture tool (its registered parameters are a placeholder) plus two listeners, registered once per root context and shared by every holder: + +- an `agent/request` waterfall listener registered `prepend: true` that post-processes `await next()` — **final-request enforcement**: the request that hits the wire never carries `structured_output` for an agent without a structured run, and for one that has it always carries the run's OWN schema (as the tool's `parameters`) plus the calling instruction appended to its `system` text (the demand travels with the tool — `AgentOptions` has no per-agent prompt field to carry it). Per-agent shaping lives here because the tool registry and prompt assembly are context-global while schemas differ per concurrent child; cooperative mutate-then-`next()` would not survive a downstream listener returning a replacement request. +- an `agent/turn-continuation` listener (also `prepend: true` — an earlier-registered force-continue listener returning without `next()` must not decide the turn before the veto runs) that stops a child's turn once its output is captured, so a successful capture doesn't buy a wasted extra model step. + +The capture tool validates each call against the run's schema (`validateStructuredValue`) — violations become an `INVALID_ARGS` isError result the model retries in-turn; a valid call records the value. + +Lifetime is refcounted with two kinds of holder: each backend acquires for its plugin lifetime (`apply`), and each structured RUN holds its own acquisition from start to settle — so unregistration can never precede a live run's settle, and the runtime disposes only when the last backend AND the last run are gone. `release()` is idempotent per acquisition. ### `depthOf(agent): number` diff --git a/packages/subagent/subagent-inprocess/package.json b/packages/subagent/subagent-inprocess/package.json index f3bd774554..ecc177162f 100644 --- a/packages/subagent/subagent-inprocess/package.json +++ b/packages/subagent/subagent-inprocess/package.json @@ -26,6 +26,8 @@ "@deepseek-ai/dsh-llm": "^0.0.1", "@deepseek-ai/dsh-session": "^0.0.1", "@deepseek-ai/dsh-subagent": "^0.0.1", + "@deepseek-ai/dsh-system-prompt": "^0.0.1", + "@deepseek-ai/dsh-tools": "^0.0.1", "cordis": "^4.0.0-rc.6" }, "devDependencies": { @@ -35,6 +37,8 @@ "@deepseek-ai/dsh-llm": "workspace:^", "@deepseek-ai/dsh-session": "workspace:^", "@deepseek-ai/dsh-subagent": "workspace:^", + "@deepseek-ai/dsh-subagent-fork": "workspace:^", + "@deepseek-ai/dsh-subagent-spawn": "workspace:^", "@deepseek-ai/dsh-system-prompt": "workspace:^", "@deepseek-ai/dsh-tools": "workspace:^", "cordis": "^4.0.0-rc.6" diff --git a/packages/subagent/subagent-inprocess/src/index.ts b/packages/subagent/subagent-inprocess/src/index.ts index 1107926aa2..381f94acba 100644 --- a/packages/subagent/subagent-inprocess/src/index.ts +++ b/packages/subagent/subagent-inprocess/src/index.ts @@ -18,7 +18,21 @@ import type { Context } from 'cordis' import { AgentId, type Agent, type AgentHandle, type AgentOptions } from '@deepseek-ai/dsh-agent' import { SessionId, type SessionEvent, type TurnEndReason } from '@deepseek-ai/dsh-session' import type { ContentBlock } from '@deepseek-ai/dsh-llm' +import { assertSupportedOutputSchema } from '@deepseek-ai/dsh-tools' import type { SubagentResult, SubagentRun, SubagentStartRequest, SubagentStopReason } from '@deepseek-ai/dsh-subagent' +import { + acquireStructuredRuntime, + STRUCTURED_OUTPUT_NUDGE, + type StructuredAcquisition, +} from './structured.ts' + +export { + acquireStructuredRuntime, + STRUCTURED_OUTPUT_TOOL, + STRUCTURED_OUTPUT_INSTRUCTION, + STRUCTURED_OUTPUT_NUDGE, + type StructuredAcquisition, +} from './structured.ts' declare module '@deepseek-ai/dsh-agent' { interface AgentOptions { @@ -76,6 +90,13 @@ export interface InProcessRunOptions { * parent's log (FORK), or `undefined` for a fresh child (SPAWN). */ readonly seed?: SessionEvent[] + /** + * How many times a structured run re-prompts a child that finished a turn + * cleanly WITHOUT calling `structured_output` (see the structured module). + * REQUIRED, resolved from the backend's validated Config — per the explicit- + * defaulting rule, the driver never fills it with a hidden fallback. + */ + readonly structuredNudgeRetries: number } /** @@ -98,6 +119,10 @@ export function startInProcessRun( if (request.maxDepth !== undefined && childDepth > request.maxDepth) { throw new SubagentDepthError(childDepth, request.maxDepth) } + // Assert the schema subset BEFORE any child exists (the service has already + // capability-gated; this rejects a schema outside the enforced subset loud). + const schema = request.outputSchema + if (schema !== undefined) assertSupportedOutputSchema(schema) const childId = AgentId(randomUUID()) // The child's OWN events begin after the seed (fork seeds the parent's @@ -109,13 +134,20 @@ export function startInProcessRun( // Inherit the parent's model by default (a child with no model cannot run); // an explicit `request.agentOptions.model` overrides it. The persona needs // no inheritance: the deployment persona is a context-wide prompt section, - // so parent and child render the same one. + // so parent and child render the same one. A structured run's + // structured_output instruction is NOT prompt state either — the structured + // runtime's final-request listener appends it per request (see structured.ts). const agentOptions: AgentOptions = { ...request.parent.options.model !== undefined ? { model: request.parent.options.model } : {}, ...request.agentOptions, subagentDepth: childDepth, } + // The structured runtime is held for the WHOLE run (acquired before the child + // exists, released when the result settles), so a backend hot-reload mid-run + // cannot unregister the capture tool out from under this live child. + const structured: StructuredAcquisition | undefined = schema !== undefined ? acquireStructuredRuntime(ctx) : undefined + const handle: AgentHandle = ctx.agents.create({ agentId: childId, sessionId: SessionId(randomUUID()), @@ -130,6 +162,7 @@ export function startInProcessRun( agentOptions, }) const child = handle.agent + if (structured && schema !== undefined) structured.attach(child, schema) // Bridge the request's abort signal to the child (the consumer also bridges // its own exec.signal, but a backend-level bridge keeps the contract local). @@ -138,6 +171,10 @@ export function startInProcessRun( // `turn/end` is logged — settles as `aborted` (honoring the cancel contract) // rather than falling through to the no-turn `error` mapping. let cancelled = false + // An accessor, not an inline read: `cancelled` mutates from closures (the + // abort listener, run.cancel), which control-flow narrowing cannot see — an + // inline `!cancelled` in the nudge condition reads as always-true. + const isCancelled = (): boolean => cancelled const requestCancel = (reason: string): void => { cancelled = true child.cancel(reason) @@ -154,9 +191,35 @@ export function startInProcessRun( if (request.signal?.aborted) return { output: [], stopReason: 'aborted' } child.send(request.prompt) await child.whenIdle() - return readResult(child, seedLength, cancelled) + if (structured) { + // Nudge loop: a child that finished a turn CLEANLY without calling + // structured_output gets re-prompted, up to the backend-configured + // retry count. An errored/aborted turn is not nudged — its failure is + // the honest result (a cancelled turn ends `aborted`, and a pre-turn + // cancel leaves no `turn/end` at all, so neither reads `completed`). + // `!cancelled` closes the remaining window: a cancel landing AFTER a + // clean turn end clears nothing — `child.cancel()` only kills + // queued/running work — so without it the next `send` would spend a + // fresh post-cancellation turn; the condition re-evaluates after + // every `whenIdle()`, so a mid-nudge cancel stops the loop at the + // next boundary too. + let nudges = options.structuredNudgeRetries + while ( + !isCancelled() && structured.captured(child) === undefined && nudges > 0 + && lastOwnTurnEnd(child, seedLength)?.data.reason.kind === 'completed' + ) { + nudges -= 1 + child.send([{ type: 'text', text: STRUCTURED_OUTPUT_NUDGE }]) + await child.whenIdle() + } + } + return readResult(child, seedLength, isCancelled(), structured ? { captured: structured.captured(child) } : undefined) } finally { request.signal?.removeEventListener('abort', onAbort) + if (structured) { + structured.detach(child) + structured.release() + } } })() @@ -173,6 +236,12 @@ export function startInProcessRun( } } +/** The child's OWN last `turn/end` event (events at or after `seedLength`), if any. */ +function lastOwnTurnEnd(child: Agent, seedLength: number): SessionEvent<'turn/end'> | undefined { + return child.session.events.slice(seedLength) + .findLast((e): e is SessionEvent<'turn/end'> => e.type === 'turn/end') +} + /** * Read a settled child's terminal result from its session log, scoped to the * child's OWN events (everything at or after `seedLength` — fork seeds the @@ -184,12 +253,32 @@ export function startInProcessRun( * logged (a cancel landed in the pre-turn window, before any turn ran), the * run settles `aborted` per the {@link SubagentRun.cancel} contract rather than * the generic no-turn `error`. + * + * A structured run (`structured` present) additionally reports the captured + * value on {@link SubagentResult.structured}. A structured child that finished + * CLEANLY without ever capturing (the nudges ran out) settles `error` — a clean + * finish without the demanded structured result is a failure, not a success + * with a missing field; a non-`completed` reason keeps its own honest mapping. */ -function readResult(child: Agent, seedLength: number, cancelled: boolean): SubagentResult { +function readResult( + child: Agent, + seedLength: number, + cancelled: boolean, + structured?: { captured?: { value: unknown } | undefined }, +): SubagentResult { const own = child.session.events.slice(seedLength) const lastMessage = own.findLast((e): e is SessionEvent<'assistant/message'> => e.type === 'assistant/message') const lastEnd = own.findLast((e): e is SessionEvent<'turn/end'> => e.type === 'turn/end') const output: ContentBlock[] = lastMessage ? structuredClone(lastMessage.data.content) : [] - if (lastEnd === undefined && cancelled) return { output, stopReason: 'aborted' } - return { output, stopReason: toStopReason(lastEnd?.data.reason) } + const stopReason: SubagentStopReason = lastEnd === undefined && cancelled + ? 'aborted' + : toStopReason(lastEnd?.data.reason) + if (structured) { + if (structured.captured) return { output, structured: structured.captured.value, stopReason } + // No capture on a cleanly-completed turn: an ERROR when the run was left + // to finish (the nudges ran out), but ABORTED when a cancel is why the + // nudging stopped — the cancel contract outranks the schema shortfall. + if (stopReason === 'completed') return { output, stopReason: cancelled ? 'aborted' : 'error' } + } + return { output, stopReason } } diff --git a/packages/subagent/subagent-inprocess/src/structured.ts b/packages/subagent/subagent-inprocess/src/structured.ts new file mode 100644 index 0000000000..69e4ea4fd6 --- /dev/null +++ b/packages/subagent/subagent-inprocess/src/structured.ts @@ -0,0 +1,239 @@ +/** + * Structured-output support for the in-process subagent backends: the mechanism + * behind `SubagentStartRequest.outputSchema` for children that run as agents on + * the same context. + * + * The model-facing surface is one globally registered `structured_output` tool + * whose REGISTERED parameters are a placeholder — the real schema is per run. + * Because the tool registry and prompt assembly are context-global while + * schemas differ per child (two concurrent structured runs may carry different + * schemas), per-agent shaping happens on the `system-prompt/assemble` + * waterfall with a `prepend: true` listener that post-processes `await next()` + * — FINAL-ASSEMBLY enforcement: whatever downstream listeners mutated or + * replaced, the assembly the loop renders never carries `structured_output` + * for an agent without a structured run, and for one that has it always + * carries the run's OWN schema plus a trailing + * {@link STRUCTURED_OUTPUT_INSTRUCTION} section (the demand travels with the + * tool). The loop logs what the assembly produced as the request header, so + * the injection is a reconstructable fact of the session log, never a + * wire-only mutation (the reconstructability RFC). + * (Cooperative mutate-then-`next()` would not survive a downstream listener + * returning a replacement assembly — see the waterfall composition caveat in + * docs/architecture.md.) + * + * A companion `agent/turn-continuation` listener stops a child's turn once its + * output is captured — without it, the loop's default "had tool calls ⇒ + * continue" buys a wasted extra model step per structured child. It is also + * `prepend: true`: the veto must run before any earlier-registered listener + * that could short-circuit the chain into a forced continue. + * + * Lifetime is refcounted with two kinds of holder: each backend acquires for + * its plugin lifetime (so the tool exists before any run), and each structured + * RUN acquires from start to settle (so a backend hot-reload mid-run cannot + * unregister the capture tool out from under a live child). Registrations are + * effects on the ROOT context — their natural upper bound is app teardown — and + * the refcount disposes them when the last holder releases. + * + * @module @deepseek-ai/dsh-subagent-inprocess/structured + */ + +import type { Context } from 'cordis' +import type { Agent } from '@deepseek-ai/dsh-agent' +import type { ContentBlock, ToolSchema } from '@deepseek-ai/dsh-llm' +import type { ContinuationDecision } from '@deepseek-ai/dsh-agent' +import type { AssembleContext, PromptAssembly } from '@deepseek-ai/dsh-system-prompt' +import type { ToolExecution } from '@deepseek-ai/dsh-tools' +import { ToolArgsError, validateStructuredValue, type StructuredOutputSchema } from '@deepseek-ai/dsh-tools' + +/** The model-facing tool name a structured child must call to finish. */ +export const STRUCTURED_OUTPUT_TOOL = 'structured_output' + +/** + * The instruction the assembly listener appends to a structured child's + * system prompt as a trailing section on every assembly. Per-assembly state, + * NOT agent prompt state: `AgentOptions` has no prompt field (the persona is + * deployment config on the system-prompt plugin), so the same final-assembly + * enforcement that injects the schema'd tool carries the instruction that + * demands calling it. + */ +export const STRUCTURED_OUTPUT_INSTRUCTION + = 'When you have your final answer, you MUST report it by calling the ' + + `\`${STRUCTURED_OUTPUT_TOOL}\` tool with arguments matching its parameter schema exactly. ` + + 'Do not finish with a plain text answer: only the tool call counts as your result.' + +/** The nudge sent when a structured child finishes cleanly without calling the tool. */ +export const STRUCTURED_OUTPUT_NUDGE + = `You finished without calling \`${STRUCTURED_OUTPUT_TOOL}\`. ` + + `Call \`${STRUCTURED_OUTPUT_TOOL}\` now with your final result matching its parameter schema.` + +/** One structured run's state: the schema to enforce and the captured value, once recorded. */ +interface RunState { + readonly schema: StructuredOutputSchema + captured?: { value: unknown } +} + +/** The per-root-context runtime: run states plus the shared registrations. */ +interface StructuredRuntime { + refs: number + readonly states: WeakMap + readonly disposers: (() => void)[] +} + +/** One root context ⇒ one runtime (multi-app test isolation). */ +const runtimes = new WeakMap() + +/** + * One holder's handle on the shared structured runtime. `release()` is + * idempotent per acquisition; the runtime's registrations are disposed when the + * LAST holder (backend plugin or live run) releases. + */ +export interface StructuredAcquisition { + /** Enforce `schema` on `agent`'s requests and start capturing its `structured_output` call. */ + attach(agent: Agent, schema: StructuredOutputSchema): void + /** The captured value, once the child called the tool with valid arguments. */ + captured(agent: Agent): { value: unknown } | undefined + /** Stop enforcing/capturing for `agent` (WeakMap-backed; safe to call twice). */ + detach(agent: Agent): void + /** Drop this holder's reference (idempotent); the last release unregisters everything. */ + release(): void +} + +/** + * Acquire the per-root-context structured runtime, registering the capture tool + * and the two waterfall listeners on the FIRST acquisition. See the module doc + * for the enforcement and lifetime design. + * @param ctx - any context of the app; the runtime keys off `ctx.root`. + * @returns this holder's handle (attach/captured/detach + idempotent release). + */ +export function acquireStructuredRuntime(ctx: Context): StructuredAcquisition { + const root: Context = ctx.root + let runtime = runtimes.get(root) + if (!runtime) { + runtime = { refs: 0, states: new WeakMap(), disposers: [] } + runtimes.set(root, runtime) + registerRuntime(root, runtime) + } + runtime.refs += 1 + + let released = false + return { + attach(agent: Agent, schema: StructuredOutputSchema): void { + runtime.states.set(agent, { schema }) + }, + captured(agent: Agent): { value: unknown } | undefined { + return runtime.states.get(agent)?.captured + }, + detach(agent: Agent): void { + runtime.states.delete(agent) + }, + release(): void { + if (released) return + released = true + runtime.refs -= 1 + if (runtime.refs > 0) return + runtimes.delete(root) + for (const dispose of runtime.disposers.splice(0)) dispose() + }, + } +} + +/** Register the capture tool + the two listeners on the root context (first acquire). */ +function registerRuntime(root: Context, runtime: StructuredRuntime): void { + // The registered parameters are a PLACEHOLDER: the request listener below + // swaps in the run's real schema per child, and strips the tool entirely for + // every agent without a structured run — so this shape is never model-visible. + // + // Registration does NOT ride on the acquiring backend's plugin-level + // `inject`: a backend that waited on `tools` would apply later than it did + // before this module existed, shifting when its PROVIDER registers — and the + // delegation tool mirrors provider lifecycle, so that shift would reorder + // the model-visible tool list of every existing prompt. Instead the capture + // tool registers synchronously when `tools` is already live (the common + // case), and through a scoped inject fiber when the Loader happens to start + // the backend first. Either way the registration lands on root and is + // disposed by the runtime's refcount; disposing the fiber also covers the + // never-activated case. + let disposeTool: (() => void) | undefined + const registerCapture = (tools: Context['tools']): void => { + disposeTool = tools.register({ + name: STRUCTURED_OUTPUT_TOOL, + description: + 'Report your final structured result. Call this exactly once, when your answer is complete; ' + + 'the arguments must match this tool\'s parameter schema exactly.', + parameters: { type: 'object', properties: {} }, + execute(args: unknown, exec: ToolExecution): Promise { + const state = exec.agent ? runtime.states.get(exec.agent) : undefined + if (!state) { + // Reachable only if a non-structured agent somehow calls the tool (the + // request listener strips it, so the model never sees it) — fail loud + // rather than capture into nowhere. + throw new Error(`${STRUCTURED_OUTPUT_TOOL} is only available to subagents started with an output schema`) + } + const violations = validateStructuredValue(state.schema, args) + // ToolArgsError → isError result with INVALID_ARGS: the model retries + // within the same turn, exactly like a schema-validated defineTool call. + if (violations.length > 0) throw new ToolArgsError(violations) + state.captured = { value: args } + return Promise.resolve([{ type: 'text', text: 'Structured output recorded.' }]) + }, + }) + } + const liveTools = root.get('tools') + const toolsFiber = liveTools ? undefined : root.inject(['tools'], (childCtx: Context) => { + registerCapture(childCtx.root.tools) + }) + if (liveTools) registerCapture(liveTools) + runtime.disposers.push(() => { + disposeTool?.() + void toolsFiber?.dispose() + }) + + // FINAL-ASSEMBLY enforcement (prepend: true = first registered = OUTERMOST + // wrapper): post-process whatever the downstream listeners and the registry + // produced, so a downstream listener returning a replacement assembly cannot + // leak the tool to other agents or erase the child's schema. The loop logs + // the rendered assembly as the step's request header, so the swap is + // reconstructable log state, never a wire-only mutation. + runtime.disposers.push(root.on('system-prompt/assemble', async function ( + this: unknown, _assembly: PromptAssembly, context: AssembleContext, next: () => Promise, + ): Promise { + const final = await next() + const state = context.agent ? runtime.states.get(context.agent) : undefined + if (state) { + const schemaEntry: ToolSchema = { + name: STRUCTURED_OUTPUT_TOOL, + description: + 'Report your final structured result. Call this exactly once, when your answer is complete; ' + + 'the arguments must match this tool\'s parameter schema exactly.', + // ToolSchema.parameters is the wire-level JSON Schema object; the + // asserted subset type is structurally exactly that. + parameters: state.schema as unknown as Record, + } + final.tools = [...final.tools.filter(tool => tool.name !== STRUCTURED_OUTPUT_TOOL), schemaEntry] + // The demand travels WITH the tool: a trailing section in the + // tool-guidance order band, appended after next() so it renders last + // (renderPrompt joins in array order). + final.sections = [...final.sections, { name: `tool:${STRUCTURED_OUTPUT_TOOL}`, order: 190, text: STRUCTURED_OUTPUT_INSTRUCTION }] + return final + } + // No structured run: strip the placeholder so it is never model-visible. + // An empty tools array canonicalizes to an absent header/wire field + // (canonicalHeader pins empty ≡ absent), so no re-shaping is needed here. + final.tools = final.tools.filter(tool => tool.name !== STRUCTURED_OUTPUT_TOOL) + return final + }, { prepend: true })) + + // Stop a structured child's turn once its output is captured: the default + // "had tool calls ⇒ continue" would otherwise buy a wasted extra model step + // after every successful capture. `prepend: true` puts the veto OUTERMOST — + // an earlier-registered listener that short-circuits the chain (a goal-style + // force-continue returning without `next()`) would otherwise decide the turn + // before this listener ever ran, and no downstream decision may resurrect a + // structured turn that is already finished. + runtime.disposers.push(root.on('agent/turn-continuation', function ( + this: unknown, agent: Agent, _turn: number, _decision: ContinuationDecision, next: () => Promise, + ): Promise { + if (runtime.states.get(agent)?.captured) return Promise.resolve({ action: 'stop' }) + return next() + }, { prepend: true })) +} diff --git a/packages/subagent/subagent-inprocess/tests/structured.spec.ts b/packages/subagent/subagent-inprocess/tests/structured.spec.ts new file mode 100644 index 0000000000..6982f7cde5 --- /dev/null +++ b/packages/subagent/subagent-inprocess/tests/structured.spec.ts @@ -0,0 +1,485 @@ +import { describe, expect, it } from 'vitest' +import { Context } from 'cordis' +import LlmService, { type GenerateOptions } from '@deepseek-ai/dsh-llm' +import SessionStore from '@deepseek-ai/dsh-session' +import SystemPrompt from '@deepseek-ai/dsh-system-prompt' +import ToolRegistry from '@deepseek-ai/dsh-tools' +import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent' +import type { Agent, ContinuationDecision } from '@deepseek-ai/dsh-agent' +import AgentLoop from '@deepseek-ai/dsh-agent-loop' +import * as Invariants from '@deepseek-ai/dsh-invariants' +import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent' +import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools' +import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts' +import * as spawn from '@deepseek-ai/dsh-subagent-spawn' +import * as fork from '@deepseek-ai/dsh-subagent-fork' +import { + acquireStructuredRuntime, + STRUCTURED_OUTPUT_INSTRUCTION, + STRUCTURED_OUTPUT_TOOL, +} from '../src/structured.ts' + +type Script = ConstructorParameters[0] + +const SCHEMA: StructuredOutputSchema = { + type: 'object', + properties: { answer: { type: 'number' }, note: { type: 'string' } }, + required: ['answer'], +} + +/** + * Real loop + scripted mock model + the REAL spawn backend (which acquires the + * structured runtime at apply, exactly as shipped). The mock model script + * drives the child's structured_output calls. + */ +async function setup(script: Script, options?: { nudges?: number; withFork?: boolean }) { + const ctx = new Context() + const adapter = new MockAdapter(script) + await ctx.plugin(LlmService) + await ctx.plugin(SessionStore) + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + await ctx.plugin(AgentRegistry) + await ctx.plugin(Invariants) + await ctx.plugin(AgentLoop, { agents: [] }) + await ctx.plugin(SubagentService) + const fiber = await ctx.plugin(spawn, { providerName: 'spawn', structuredNudgeRetries: options?.nudges ?? 1 }) + const forkFiber = options?.withFork + ? await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: options?.nudges ?? 1 }) + : undefined + ctx.llm.registerAdapter(['mock'], adapter) + const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) + return { ctx, parent, adapter, fiber, forkFiber } +} + +function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial): SubagentStartRequest { + return { prompt: [{ type: 'text', text: 'produce the answer' }], parent, outputSchema: SCHEMA, ...extra } +} + +/** The tool names of one recorded model request. */ +function toolNames(request: GenerateOptions): string[] { + return (request.tools ?? []).map(tool => tool.name) +} + +describe('in-process structured output', () => { + it('captures a valid structured_output call and surfaces result.structured', async () => { + const { ctx, parent } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }), + ]) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.stopReason).toBe('completed') + expect(result.structured).toEqual({ answer: 42, note: 'done' }) + await run.dispose() + }) + + it('stops the turn after a successful capture — no extra model step is spent', async () => { + const { ctx, parent, adapter } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }), + textResponse('MUST NOT BE CONSUMED'), + ]) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + await run.result + // Default continuation would run a second step after the tool call; the + // structured runtime's turn-continuation veto stops the turn instead. + expect(adapter.requests.length).toBe(1) + await run.dispose() + }) + + it('the captured-turn veto is prepend: an EARLIER force-continue listener cannot short-circuit it', async () => { + const ctx = new Context() + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + // Registered BEFORE the structured runtime exists — without prepend, this + // goal-style listener would decide the turn first (returning WITHOUT + // calling next()) and the veto would never run. + ctx.on('agent/turn-continuation', () => Promise.resolve({ action: 'continue' })) + const acquisition = acquireStructuredRuntime(ctx) + const agent = { id: AgentId('structured-child') } as unknown as Agent + acquisition.attach(agent, SCHEMA) + const captured = await ctx.tools.execute({ + callId: 'call-1' as never, + name: STRUCTURED_OUTPUT_TOOL, + arguments: { answer: 1 }, + agent, + }) + expect(captured.isError).toBeFalsy() + const decision = await ctx.waterfall( + 'agent/turn-continuation', agent, 1, + { action: 'continue' }, + () => Promise.resolve({ action: 'continue' }), + ) + expect(decision).toEqual({ action: 'stop' }) + acquisition.detach(agent) + acquisition.release() + }) + + it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => { + const { ctx, parent } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }), + toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }), + ]) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.structured).toEqual({ answer: 7 }) + expect(result.stopReason).toBe('completed') + // The child's log carries the isError tool/result for the invalid call. + const child = ctx.agents.get(run.id)! + const results = child.session.events.filter(e => e.type === 'tool/result') + expect(results.length).toBe(2) + expect((results[0]!.data as { isError?: boolean }).isError).toBe(true) + await run.dispose() + }) + + it('nudges a child that finished cleanly without calling the tool, then captures', async () => { + const { ctx, parent } = await setup([ + textResponse('here is my answer in prose'), + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }), + ]) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.structured).toEqual({ answer: 3 }) + expect(result.stopReason).toBe('completed') + // The nudge is a real user-visible message in the child's log. + const child = ctx.agents.get(run.id)! + const users = child.session.events.filter(e => e.type === 'user/message') + expect(users.length).toBe(2) + await run.dispose() + }) + + it('settles error when the nudges run out without a capture', async () => { + const { ctx, parent, adapter } = await setup([ + textResponse('prose only'), + textResponse('still prose'), + ], { nudges: 1 }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.stopReason).toBe('error') + expect(result.structured).toBeUndefined() + expect(adapter.requests.length).toBe(2) + await run.dispose() + }) + + it('zero nudge retries fails immediately after the first clean prose finish', async () => { + const { ctx, parent, adapter } = await setup([textResponse('prose')], { nudges: 0 }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.stopReason).toBe('error') + expect(adapter.requests.length).toBe(1) + await run.dispose() + }) + + it('a child that errored is NOT nudged (its failure is the honest result)', async () => { + // Script exhaustion on the first call → the child turn errors. + const { ctx, parent, adapter } = await setup([], { nudges: 3 }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.stopReason).toBe('error') + expect(adapter.requests.length).toBe(1) + await run.dispose() + }) + + it('a cancel landing after a clean turn end stops the nudge loop: no post-cancellation turn is spent', async () => { + const { ctx, parent, adapter } = await setup([textResponse('prose, no capture')], { nudges: 3 }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const child = ctx.agents.get(run.id)! + // Cancel synchronously inside the first turn's end recording — after the + // turn reads `completed`, before the nudge continuation resumes. The turn + // state alone cannot see this cancel (`child.cancel()` only clears + // queued/running work), so without the loop's own cancelled check the + // next send would spend a fresh child turn after the caller cancelled. + ctx.on('session/event', (session, event) => { + if (session === child.session && event.type === 'turn/end') run.cancel('cancelled between turn end and nudge') + }) + const result = await run.result + expect(result.stopReason).toBe('aborted') + // Exactly one model request: the nudge turn never ran. + expect(adapter.requests.length).toBe(1) + await run.dispose() + }) + + it('rejects a schema outside the subset loud, before any child exists', async () => { + const { ctx, parent } = await setup([]) + expect(() => ctx.subagents.start('spawn', structuredRequest(parent, { + outputSchema: { type: 'object', oneOf: [] } as unknown as StructuredOutputSchema, + }))).toThrow(/unsupported output schema/) + expect(ctx.agents.get(AgentId('parent'))).toBeDefined() + }) + + it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => { + const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })]) + // A context-wide section stands in for the deployment persona: the + // instruction must APPEND to whatever the prompt pipeline assembled, not + // replace it (AgentOptions has no prompt field — the instruction is + // per-request wire state added by the final-request listener). + ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + await run.result + const childRequest = adapter.requests.at(-1)! + expect(childRequest.system).toContain('You are a counter.') + expect(childRequest.system!.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true) + expect(childRequest.system!.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0) + await run.dispose() + }) + + it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => { + const { ctx, parent, adapter } = await setup([ + textResponse('parent answer'), + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }), + ]) + parent.send([{ type: 'text', text: 'hello' }]) + await parent.whenIdle() + expect(adapter.requests[0]!.system ?? '').not.toContain(STRUCTURED_OUTPUT_INSTRUCTION) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + await run.result + // The loop always assembles a base prompt (the harness identity section), + // so the instruction APPENDS — never replaces. + const childSystem = adapter.requests.at(-1)!.system! + expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true) + expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length) + await run.dispose() + }) + + describe('final-request enforcement (the prepend agent/request listener)', () => { + it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => { + const { ctx, parent, adapter } = await setup([ + // Parent turn (a plain agent): must NOT see the tool. + textResponse('parent answer'), + // Child turn: must see it, with the run's schema. + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }), + ]) + parent.send([{ type: 'text', text: 'hello' }]) + await parent.whenIdle() + expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL) + + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + await run.result + const childRequest = adapter.requests[1]! + expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL) + const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)! + expect(entry.parameters).toEqual(SCHEMA) + await run.dispose() + }) + + it('two concurrent structured children each see their OWN schema', async () => { + const otherSchema: StructuredOutputSchema = { + type: 'object', + properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } }, + required: ['verdict'], + } + const { ctx, parent, adapter } = await setup([ + (options: GenerateOptions) => { + // Answer with whatever schema this child was given — proves each + // request carried the right one regardless of scheduling order. + const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)! + const args = 'verdict' in (entry.parameters.properties as Record) + ? { verdict: 'real' } + : { answer: 1 } + return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args) + }, + (options: GenerateOptions) => { + const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)! + const args = 'verdict' in (entry.parameters.properties as Record) + ? { verdict: 'real' } + : { answer: 1 } + return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args) + }, + ]) + const runA = ctx.subagents.start('spawn', structuredRequest(parent)) + const runB = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema })) + const [a, b] = await Promise.all([runA.result, runB.result]) + expect(a.structured).toEqual({ answer: 1 }) + expect(b.structured).toEqual({ verdict: 'real' }) + const schemas = adapter.requests.map(request => + request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters) + expect(schemas).toContainEqual(SCHEMA) + expect(schemas).toContainEqual(otherSchema) + await runA.dispose() + await runB.dispose() + }) + + it('wins against a downstream listener that REPLACES the assembly object', async () => { + const { ctx, parent, adapter } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }), + ]) + // A downstream (non-prepend) listener that returns a brand-new assembly — + // the composition caveat that erases cooperative mutations. Registered + // AFTER the runtime's prepend listener, so it runs INSIDE it. + ctx.on('system-prompt/assemble', async (_assembly, _context, next) => { + const replaced = await next() + return { sections: [...replaced.sections], tools: [...replaced.tools], variables: { ...replaced.variables } } + }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.structured).toEqual({ answer: 5 }) + const entry = adapter.requests[0]!.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL) + expect(entry).toBeDefined() + expect(entry!.parameters).toEqual(SCHEMA) + await run.dispose() + }) + + it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => { + const { parent, adapter } = await setup([ + // The registry contributes the placeholder via prompt assembly, so + // tools is an array in the raw request — but after stripping the + // placeholder (its ONLY entry), the field must not be re-added as a + // different shape. + textResponse('plain'), + ]) + parent.send([{ type: 'text', text: 'q' }]) + await parent.whenIdle() + const request = adapter.requests[0]! + expect(toolNames(request)).not.toContain(STRUCTURED_OUTPUT_TOOL) + await new Promise(resolve => setTimeout(resolve, 0)) + }) + + it('shapes a bare assembly on the waterfall: no-agent context strips the placeholder; a structured agent gains schema + trailing instruction section', async () => { + // Drive ctx.systemPrompt.assemble directly — the enforcement listener + // must tolerate a context with NO agent (a bare diagnostic assemble) + // and shape a structured agent's assembly on the same path the loop + // renders and logs as the request header. + const { ctx, parent } = await setup([]) + const bare = await ctx.systemPrompt.assemble({}) + expect(bare.tools.map(tool => tool.name)).not.toContain(STRUCTURED_OUTPUT_TOOL) + + const acquisition = acquireStructuredRuntime(ctx) + acquisition.attach(parent, SCHEMA) + const shaped = await ctx.systemPrompt.assemble({ agent: parent }) + expect(shaped.tools.map(tool => tool.name)).toContain(STRUCTURED_OUTPUT_TOOL) + expect(shaped.tools.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters).toEqual(SCHEMA) + // The demand travels with the tool: the instruction renders LAST + // (appended post-next(); renderPrompt joins in array order). + expect(shaped.sections.at(-1)).toMatchObject({ name: `tool:${STRUCTURED_OUTPUT_TOOL}`, text: STRUCTURED_OUTPUT_INSTRUCTION }) + acquisition.detach(parent) + acquisition.release() + }) + }) + + describe('runtime lifetime (refcount: backends + live runs)', () => { + it('registers the capture tool while a backend is loaded and unregisters when the last unloads', async () => { + const { ctx, fiber, forkFiber } = await setup([], { withFork: true }) + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + await fiber.dispose() + // fork still holds a reference. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + await forkFiber!.dispose() + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + }) + + it('a live run-level acquisition keeps the runtime registered after EVERY backend unloads', async () => { + // Simulates the run-holder half of the two-level lifetime: a structured + // run acquires at start and releases at settle, so registration ordering + // is settle-then-unregister even if all backends unload first. (A real + // in-process child dies WITH its backend's fiber — the acquisition's + // observable job is this ordering, which a manual holder pins directly.) + const { ctx, fiber, forkFiber } = await setup([], { withFork: true }) + const runHolder = acquireStructuredRuntime(ctx) + await fiber.dispose() + await forkFiber!.dispose() + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + runHolder.release() + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + }) + + it('a structured run releases its acquisition when it settles (backend unload mid-run)', async () => { + const { ctx, parent, fiber } = await setup(['hang']) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + // Let the child's step start streaming, then unload the backend. The + // backend owns the child agent, so the unload tears the child down and + // the run settles — releasing its own acquisition on the way out. + await new Promise(resolve => setTimeout(resolve, 30)) + await fiber.dispose() + const result = await run.result + expect(result.stopReason).toBe('error') + // Both holders (backend + run) released — nothing keeps the runtime now. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + await run.dispose() + }) + + it('fork children capture structured output through the same runtime', async () => { + const { ctx, parent } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }), + ], { withFork: true }) + const run = ctx.subagents.start('fork', structuredRequest(parent)) + const result = await run.result + expect(result.structured).toEqual({ answer: 9 }) + await run.dispose() + }) + + it('acquisition release is idempotent (double release cannot underflow the refcount)', async () => { + const ctx = new Context() + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + const first = acquireStructuredRuntime(ctx) + const second = acquireStructuredRuntime(ctx) + first.release() + first.release() + // The second holder still keeps the tool registered. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + second.release() + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + }) + + it('registers the capture tool through the scoped fiber when tools loads after the acquisition', async () => { + // The Loader starts sibling plugins concurrently, so a backend can + // acquire the runtime before dsh-tools has applied. The capture tool + // must then register as soon as `tools` exists — via the inject fiber, + // not by deferring the backend (which would reorder the prompt's tools). + const ctx = new Context() + const acquisition = acquireStructuredRuntime(ctx) + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + // Fiber activation completes asynchronously after the service appears. + await new Promise(resolve => setImmediate(resolve)) + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + acquisition.release() + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + }) + + it('releasing before tools ever loads disposes the pending fiber without registering', async () => { + const ctx = new Context() + const acquisition = acquireStructuredRuntime(ctx) + acquisition.release() + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + await new Promise(resolve => setImmediate(resolve)) + // The disposed fiber never fires: nothing registers after the fact. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + }) + + it('attach/captured/detach manage per-agent state through the acquisition surface', async () => { + const { ctx, parent } = await setup([]) + const acquisition = acquireStructuredRuntime(ctx) + expect(acquisition.captured(parent)).toBeUndefined() + acquisition.attach(parent, SCHEMA) + expect(acquisition.captured(parent)).toBeUndefined() + acquisition.detach(parent) + acquisition.detach(parent) + acquisition.release() + // The backend still holds its own reference from setup(). + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + }) + }) + + it('a direct structured_output call from an agent WITHOUT a structured run is an isError', async () => { + const { ctx, parent } = await setup([]) + const result = await ctx.tools.execute({ + callId: 'x' as never, + name: STRUCTURED_OUTPUT_TOOL, + arguments: { answer: 1 }, + agent: parent, + }) + expect(result.isError).toBe(true) + expect(result.content[0]).toMatchObject({ type: 'text' }) + }) + + it('a structured_output call with NO calling agent at all is an isError', async () => { + const { ctx } = await setup([]) + const result = await ctx.tools.execute({ + callId: 'x' as never, + name: STRUCTURED_OUTPUT_TOOL, + arguments: { answer: 1 }, + }) + expect(result.isError).toBe(true) + }) +}) diff --git a/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts b/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts index 7219e03988..d3870ae51a 100644 --- a/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts @@ -51,7 +51,7 @@ describe('depthOf', () => { describe('startInProcessRun', () => { it('drives a fresh child (no seed) to completion and returns its output', async () => { const { ctx, parent } = await setup([textResponse('driver child answer')]) - const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'do X' }], parent }, { providerName: 'spawn' }) + const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'do X' }], parent }, { providerName: 'spawn', structuredNudgeRetries: 1 }) const result = await run.result expect(result.stopReason).toBe('completed') expect(text(result.output)).toBe('driver child answer') @@ -61,7 +61,7 @@ describe('startInProcessRun', () => { it('throws SubagentDepthError when the child would exceed maxDepth', async () => { const { ctx, parent } = await setup([]) - expect(() => startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'p' }], parent, maxDepth: 0 }, { providerName: 'spawn' })) + expect(() => startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'p' }], parent, maxDepth: 0 }, { providerName: 'spawn', structuredNudgeRetries: 1 })) .toThrow(SubagentDepthError) }) @@ -73,7 +73,7 @@ describe('startInProcessRun', () => { parent.send([{ type: 'text', text: 'parent q' }]) await parent.whenIdle() const seed = parent.session.events.slice() - const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'child q' }], parent }, { providerName: 'fork', seed }) + const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'child q' }], parent }, { providerName: 'fork', structuredNudgeRetries: 1, seed }) const result = await run.result expect(result.stopReason).toBe('completed') expect(text(result.output)).toBe('seeded child reply') diff --git a/packages/subagent/subagent-inprocess/tsconfig.json b/packages/subagent/subagent-inprocess/tsconfig.json index 4cb435d4fb..7b7a015cc9 100644 --- a/packages/subagent/subagent-inprocess/tsconfig.json +++ b/packages/subagent/subagent-inprocess/tsconfig.json @@ -25,6 +25,12 @@ }, { "path": "../subagent" + }, + { + "path": "../../core/system-prompt" + }, + { + "path": "../../core/tools" } ] } diff --git a/packages/subagent/subagent-spawn/README.md b/packages/subagent/subagent-spawn/README.md index 97dfae9304..059c996215 100644 --- a/packages/subagent/subagent-spawn/README.md +++ b/packages/subagent/subagent-spawn/README.md @@ -6,14 +6,15 @@ The run mechanics live in the shared [`@deepseek-ai/dsh-subagent-inprocess`](../ ## What it does -`start(request)` delegates to `startInProcessRun(ctx, request, { providerName })` with no seed: a fresh child agent with the parent's `cwd`/`parentSession` lineage and (by default) the parent's model. See the [driver README](../subagent-inprocess/README.md) for the full lifecycle (depth check, one-shot drive, result read, dispose). +`start(request)` delegates to `startInProcessRun(ctx, request, { providerName, structuredNudgeRetries })` with no seed: a fresh child agent with the parent's `cwd`/`parentSession` lineage and (by default) the parent's model. See the [driver README](../subagent-inprocess/README.md) for the full lifecycle (depth check, one-shot drive, result read, dispose). ## Capabilities -`{ outputSchema: false, depthLimit: true, toolFilter: false }`. It constructs the child, so it enforces a recursion cap; structured output and tool-scoping are deferred (the service rejects a request needing either before `start` runs). +`{ outputSchema: true, depthLimit: true, toolFilter: false }`. It constructs the child, so it enforces a recursion cap, and it supports structured output via the driver's shared [structured runtime](../subagent-inprocess/README.md) (the backend acquires it for its plugin lifetime; each structured run holds its own acquisition until it settles). Tool-scoping is deferred (the service rejects a request needing it before `start` runs). ## Config | Key | Meaning | |---|---| | `providerName` | Registry name on `ctx.subagents` (default `spawn`). | +| `structuredNudgeRetries` | How many times a structured run re-prompts a child that finished cleanly without calling `structured_output` (default 1). | diff --git a/packages/subagent/subagent-spawn/src/index.ts b/packages/subagent/subagent-spawn/src/index.ts index e6cf5039a7..248e887ce7 100644 --- a/packages/subagent/subagent-spawn/src/index.ts +++ b/packages/subagent/subagent-spawn/src/index.ts @@ -9,6 +9,11 @@ * ({@link startInProcessRun}); this backend just passes NO seed (a fresh * child). The fork backend is an independent peer over the same driver. * + * Structured output (`outputSchema`) is supported via the driver's shared + * structured runtime: the backend acquires it for its plugin lifetime (so the + * capture tool and request-shaping listeners exist before any run), and each + * structured run holds its own acquisition until it settles. + * * Plugin export shape: named `name`/`inject`/`Config`/`apply`, NO default. * * @module @deepseek-ai/dsh-subagent-spawn @@ -17,40 +22,68 @@ import type { Context } from 'cordis' import z from 'schemastery' import type { SubagentCapabilities, SubagentProvider, SubagentStartRequest } from '@deepseek-ai/dsh-subagent' -import { startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' +import { acquireStructuredRuntime, startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' export const name = 'subagent-spawn' +// `tools` is deliberately NOT injected: the structured runtime gates its own +// capture-tool registration on `tools` availability internally, so this +// backend's apply timing — and with it the provider-mirroring delegation +// tool's position in the model-visible tool list — stays what it was before +// structured output existed. export const inject = ['subagents', 'agents'] -/** Config: the registry name to register the provider under. */ +/** Config: the registry name to register the provider under, plus structured-run tuning. */ export interface Config { /** Provider name on `ctx.subagents` (default `spawn`). */ providerName: string + /** + * How many times a structured run re-prompts a child that finished cleanly + * without calling `structured_output` before giving up (default 1). + */ + structuredNudgeRetries: number } export const Config: z = z.object({ providerName: z.string().default('spawn'), + structuredNudgeRetries: z.natural().default(1), }) /** * The spawn provider. Supports `depthLimit` (it constructs the child, so it can - * enforce a recursion cap) but NOT `outputSchema` or `toolFilter` in this cut — - * a request that needs either is rejected by the service before `start` runs. + * enforce a recursion cap) and `outputSchema` (via the shared in-process + * structured runtime); NOT `toolFilter` in this cut — a request that needs it + * is rejected by the service before `start` runs. */ class SpawnProvider implements SubagentProvider { - readonly capabilities: SubagentCapabilities = { outputSchema: false, depthLimit: true, toolFilter: false } + readonly capabilities: SubagentCapabilities = { outputSchema: true, depthLimit: true, toolFilter: false } // Context contract: a spawned child starts fresh — it never sees the parent conversation. readonly inheritsParentContext = false - constructor(readonly name: string, private readonly ctx: Context) {} + constructor( + readonly name: string, + private readonly ctx: Context, + private readonly structuredNudgeRetries: number, + ) {} start(request: SubagentStartRequest) { // Fresh child: no seed. The shared driver mints ids, stamps cwd/lineage/ - // depth, drives the one-shot, and maps the result. - return startInProcessRun(this.ctx, request, { providerName: this.name }) + // depth, drives the one-shot (including the structured capture/nudge loop + // when the request carries an outputSchema), and maps the result. + return startInProcessRun(this.ctx, request, { + providerName: this.name, + structuredNudgeRetries: this.structuredNudgeRetries, + }) } } export function apply(ctx: Context, config: Config): void { - ctx.subagents.registerProvider(new SpawnProvider(config.providerName, ctx)) + // Hold the structured runtime for the plugin's lifetime, so the capture tool + // and its request-shaping listeners are registered before the first + // structured run and torn down when the last backend unloads (live runs hold + // their own acquisitions, so an unload mid-run cannot strand a child). + ctx.effect(() => { + const acquisition = acquireStructuredRuntime(ctx) + return () => { acquisition.release() } + }, 'subagent-spawn structured runtime') + ctx.subagents.registerProvider(new SpawnProvider(config.providerName, ctx, config.structuredNudgeRetries)) } diff --git a/packages/subagent/subagent-spawn/tests/harness.ts b/packages/subagent/subagent-spawn/tests/harness.ts index b3e9d4ec24..97dab3c3ee 100644 --- a/packages/subagent/subagent-spawn/tests/harness.ts +++ b/packages/subagent/subagent-spawn/tests/harness.ts @@ -34,7 +34,7 @@ export async function spawnHarness(workdir: string): Promise { await ctx.plugin(LocalBashExecutor, { cwd: workdir, timeoutMs: 30_000 }) await ctx.plugin(ToolBash) await ctx.plugin(SubagentService) - await ctx.plugin(Spawn, { providerName: 'spawn' }) + await ctx.plugin(Spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) // The model-facing subagent tool, bound to the spawn backend. await ctx.plugin(ToolSubagent, { provider: 'spawn' }) return ctx diff --git a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts index ccfd6492f4..70ce7a4774 100644 --- a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts +++ b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts @@ -34,7 +34,7 @@ async function setup(script: Script) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - await ctx.plugin(spawn, { providerName: 'spawn' }) + await ctx.plugin(spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) ctx.llm.registerAdapter(['mock'], adapter) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) return { ctx, parent, adapter } @@ -241,17 +241,23 @@ describe('dsh-subagent-spawn', () => { await parentHandle.dispose() }) - it('advertises depthLimit but not outputSchema/toolFilter', async () => { + it('advertises depthLimit and outputSchema but not toolFilter', async () => { const { ctx } = await setup([]) const provider = ctx.subagents.getProvider('spawn')! - expect(provider.capabilities).toEqual({ outputSchema: false, depthLimit: true, toolFilter: false }) + expect(provider.capabilities).toEqual({ outputSchema: true, depthLimit: true, toolFilter: false }) }) it('unregisters the provider when its fiber is disposed (HMR safety)', async () => { const ctx = new Context() await ctx.plugin(SubagentService) await ctx.plugin(AgentRegistry) - const fiber = await ctx.plugin(spawn, { providerName: 'spawn' }) + // The backend does NOT inject 'tools' (the structured runtime gates its + // capture-tool registration on tools availability itself, keeping backend + // apply timing — and the delegation tool's prompt position — unchanged); + // the registries are loaded here so the runtime registers eagerly anyway. + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + const fiber = await ctx.plugin(spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) expect(ctx.subagents.list()).toEqual(['spawn']) await fiber.dispose() expect(ctx.subagents.list()).toEqual([]) diff --git a/packages/subagent/subagent/src/types.ts b/packages/subagent/subagent/src/types.ts index ef76a96e5d..82fd12af36 100644 --- a/packages/subagent/subagent/src/types.ts +++ b/packages/subagent/subagent/src/types.ts @@ -8,7 +8,7 @@ import type { Agent, AgentId, AgentOptions } from '@deepseek-ai/dsh-agent' import type { ContentBlock } from '@deepseek-ai/dsh-llm' -import type { SchemaSpec } from '@deepseek-ai/dsh-tools' +import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools' /** * Which START-TIME features a provider supports. Checked by the service @@ -56,12 +56,16 @@ export interface SubagentStartRequest { /** Per-child agent options (model, system prompt). */ agentOptions?: AgentOptions /** - * Optional structured-output schema. When set AND the provider's - * {@link SubagentCapabilities.outputSchema} is `true`, the child's final - * answer is shaped to this schema and surfaced as {@link SubagentResult.structured}. + * Optional structured-output schema — an object-rooted JSON Schema within the + * enforced subset (see `assertSupportedOutputSchema` in dsh-tools; a schema + * outside the subset is rejected loud at start). When set AND the provider's + * {@link SubagentCapabilities.outputSchema} is `true`, the child is driven to + * report a value matching this schema, surfaced as + * {@link SubagentResult.structured}. The schema must be plain host-realm JSON + * data — a caller holding foreign-realm data materializes it first. * Requesting it against a provider that lacks the capability is rejected at start. */ - outputSchema?: SchemaSpec + outputSchema?: StructuredOutputSchema /** * Optional recursion cap (max delegation depth below this child). Requires * {@link SubagentCapabilities.depthLimit}; rejected at start otherwise. diff --git a/packages/subagent/subagent/tests/service.spec.ts b/packages/subagent/subagent/tests/service.spec.ts index e49b4c12e0..f70abf7e72 100644 --- a/packages/subagent/subagent/tests/service.spec.ts +++ b/packages/subagent/subagent/tests/service.spec.ts @@ -178,7 +178,7 @@ describe('SubagentService', () => { describe('start-time capability validation (fail loud, before any child)', () => { it.each([ - { field: 'outputSchema', request: baseRequest({ outputSchema: { x: { type: 'string' } } }) }, + { field: 'outputSchema', request: baseRequest({ outputSchema: { type: 'object', properties: { x: { type: 'string' } } } }) }, { field: 'maxDepth', request: baseRequest({ maxDepth: 2 }) }, { field: 'toolFilter', request: baseRequest({ toolFilter: { deny: ['bash'] } }) }, ])('rejects $field against a provider that lacks the capability — before start() runs', ({ request }) => { @@ -203,7 +203,7 @@ describe('SubagentService', () => { await ctx.plugin(SubagentService) const provider = new StubProvider('strong', ALL_CAPS) ctx.subagents.registerProvider(provider) - ctx.subagents.start('strong', baseRequest({ outputSchema: { x: { type: 'string' } }, maxDepth: 1 })) + ctx.subagents.start('strong', baseRequest({ outputSchema: { type: 'object', properties: { x: { type: 'string' } } }, maxDepth: 1 })) expect(provider.startCount).toBe(1) }) }) diff --git a/packages/support/subagent-mock/tests/subagent-mock.spec.ts b/packages/support/subagent-mock/tests/subagent-mock.spec.ts index f35ed884eb..ddd725da4b 100644 --- a/packages/support/subagent-mock/tests/subagent-mock.spec.ts +++ b/packages/support/subagent-mock/tests/subagent-mock.spec.ts @@ -41,13 +41,13 @@ describe('dsh-subagent-mock', () => { it('surfaces a structured result when the request carries an outputSchema', async () => { const ctx = await mount({ reply: 'r', structured: { answer: 42 } }) - const run = ctx.subagents.start('mock', baseRequest({ outputSchema: { answer: { type: 'number' } } })) + const run = ctx.subagents.start('mock', baseRequest({ outputSchema: { type: 'object', properties: { answer: { type: 'number' } } } })) await expect(run.result).resolves.toMatchObject({ structured: { answer: 42 } }) }) it('defaults structured output to { reply } when outputSchema is requested but no structured value is configured', async () => { const ctx = await mount({ reply: 'fallback reply' }) - const run = ctx.subagents.start('mock', baseRequest({ outputSchema: { answer: { type: 'number' } } })) + const run = ctx.subagents.start('mock', baseRequest({ outputSchema: { type: 'object', properties: { answer: { type: 'number' } } } })) await expect(run.result).resolves.toMatchObject({ structured: { reply: 'fallback reply' } }) }) diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index d5fb68c746..35d940028b 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -655,6 +655,12 @@ importers: '@deepseek-ai/dsh-subagent': specifier: workspace:^ version: link:../subagent + '@deepseek-ai/dsh-subagent-fork': + specifier: workspace:^ + version: link:../subagent-fork + '@deepseek-ai/dsh-subagent-spawn': + specifier: workspace:^ + version: link:../subagent-spawn '@deepseek-ai/dsh-system-prompt': specifier: workspace:^ version: link:../../core/system-prompt From a520965f09a9721ccfec1dbaf2515fd4674ac872 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 00:11:55 +0800 Subject: [PATCH 02/24] fix review findings: post-capture tool calls denied; schema snapshotted at start MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two bot findings on the structured runtime: - Terminal means terminal WITHIN the step: a model response listing structured_output before further tool calls executed those calls after the final answer was accepted (the turn-continuation veto only fires at step end). A third runtime listener now denies every call for a captured agent at the tools/pre-execute gate — dispatch skipped, isError result naming the contract. Calls preceding the capture in the same response are untouched. - The output schema is structuredClone'd before the subset assertion: the caller keeps its reference, so asserting and attaching the original let a post-start() mutation drift the enforced schema away from the asserted one. The clone pins assertion, model-visible parameters, and validation to one value. --- .../subagent/subagent-inprocess/src/index.ts | 11 ++- .../subagent-inprocess/src/structured.ts | 31 ++++++- .../tests/structured.spec.ts | 86 ++++++++++++++++++- 3 files changed, 122 insertions(+), 6 deletions(-) diff --git a/packages/subagent/subagent-inprocess/src/index.ts b/packages/subagent/subagent-inprocess/src/index.ts index 381f94acba..1309556f3f 100644 --- a/packages/subagent/subagent-inprocess/src/index.ts +++ b/packages/subagent/subagent-inprocess/src/index.ts @@ -119,9 +119,14 @@ export function startInProcessRun( if (request.maxDepth !== undefined && childDepth > request.maxDepth) { throw new SubagentDepthError(childDepth, request.maxDepth) } - // Assert the schema subset BEFORE any child exists (the service has already - // capability-gated; this rejects a schema outside the enforced subset loud). - const schema = request.outputSchema + // Snapshot, then assert, the schema subset BEFORE any child exists (the + // service has already capability-gated; this rejects a schema outside the + // enforced subset loud). The snapshot is load-bearing: the caller keeps its + // reference, so validating and attaching the ORIGINAL would let a + // post-start() mutation drift the enforced schema away from the asserted + // one — the clone pins assertion, the model-visible parameters, and + // validateStructuredValue to the same isolation-immutable value. + const schema = request.outputSchema === undefined ? undefined : structuredClone(request.outputSchema) if (schema !== undefined) assertSupportedOutputSchema(schema) const childId = AgentId(randomUUID()) diff --git a/packages/subagent/subagent-inprocess/src/structured.ts b/packages/subagent/subagent-inprocess/src/structured.ts index 69e4ea4fd6..e866e5e699 100644 --- a/packages/subagent/subagent-inprocess/src/structured.ts +++ b/packages/subagent/subagent-inprocess/src/structured.ts @@ -25,7 +25,11 @@ * output is captured — without it, the loop's default "had tool calls ⇒ * continue" buys a wasted extra model step per structured child. It is also * `prepend: true`: the veto must run before any earlier-registered listener - * that could short-circuit the chain into a forced continue. + * that could short-circuit the chain into a forced continue. A third listener + * closes the within-step window the continuation veto cannot: a + * `tools/pre-execute` deny for any call arriving after the agent's capture, so + * a response that lists `structured_output` before further tool calls cannot + * run side effects after the final answer was accepted. * * Lifetime is refcounted with two kinds of holder: each backend acquires for * its plugin lifetime (so the tool exists before any run), and each structured @@ -42,7 +46,7 @@ import type { Agent } from '@deepseek-ai/dsh-agent' import type { ContentBlock, ToolSchema } from '@deepseek-ai/dsh-llm' import type { ContinuationDecision } from '@deepseek-ai/dsh-agent' import type { AssembleContext, PromptAssembly } from '@deepseek-ai/dsh-system-prompt' -import type { ToolExecution } from '@deepseek-ai/dsh-tools' +import type { PreToolDecision, ToolExecution } from '@deepseek-ai/dsh-tools' import { ToolArgsError, validateStructuredValue, type StructuredOutputSchema } from '@deepseek-ai/dsh-tools' /** The model-facing tool name a structured child must call to finish. */ @@ -236,4 +240,27 @@ function registerRuntime(root: Context, runtime: StructuredRuntime): void { if (runtime.states.get(agent)?.captured) return Promise.resolve({ action: 'stop' }) return next() }, { prepend: true })) + + // Terminal means terminal WITHIN the step, not only at its end: the + // turn-continuation veto above runs after every call in the current model + // response has executed, so a response that puts `structured_output` before + // further tool calls would still perform those side effects after the final + // answer was accepted. Deny every later call for a captured agent at the + // allow/deny gate — dispatch is skipped and the model sees an `isError` + // result naming the contract. Calls that PRECEDE the capture in the same + // response ran before `captured` was set and are untouched; a second + // `structured_output` is denied like any other call. `prepend: true` for the + // same reason as the continuation veto: no earlier-registered allow may + // short-circuit past the terminal contract. + runtime.disposers.push(root.on('tools/pre-execute', function ( + this: unknown, exec: ToolExecution, next: () => Promise, + ): Promise { + if (exec.agent && runtime.states.get(exec.agent)?.captured) { + return Promise.resolve({ + kind: 'deny', + reason: `structured output already recorded: the run is complete, so \`${exec.name}\` is not executed`, + }) + } + return next() + }, { prepend: true })) } diff --git a/packages/subagent/subagent-inprocess/tests/structured.spec.ts b/packages/subagent/subagent-inprocess/tests/structured.spec.ts index 6982f7cde5..006fcd9d8a 100644 --- a/packages/subagent/subagent-inprocess/tests/structured.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/structured.spec.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from 'vitest' import { Context } from 'cordis' -import LlmService, { type GenerateOptions } from '@deepseek-ai/dsh-llm' +import LlmService, { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm' import SessionStore from '@deepseek-ai/dsh-session' import SystemPrompt from '@deepseek-ai/dsh-system-prompt' import ToolRegistry from '@deepseek-ai/dsh-tools' @@ -86,6 +86,90 @@ describe('in-process structured output', () => { await run.dispose() }) + it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => { + // One model response carrying structured_output FIRST and a side-effecting + // call after it: the continuation veto only fires at step end, so without + // the pre-execute deny the trailing call would still run after the final + // answer was accepted. + const response = [ + ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2), + { type: 'block-start', index: 1, blockType: 'tool-call' }, + { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } }, + { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } }, + { type: 'finish', reason: { kind: 'tool-calls' } }, + ] as Script[number] + const { ctx, parent } = await setup([response]) + let sideEffectRan = false + ctx.tools.register({ + name: 'side_effect', + description: 'probe', + parameters: { type: 'object', properties: {} }, + execute(): Promise { + sideEffectRan = true + return Promise.resolve([{ type: 'text', text: 'ran' }]) + }, + }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.stopReason).toBe('completed') + expect(result.structured).toEqual({ answer: 5 }) + // The deny skipped dispatch entirely: the probe body never ran. + expect(sideEffectRan).toBe(false) + await run.dispose() + }) + + it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => { + const response = [ + { type: 'block-start', index: 0, blockType: 'tool-call' }, + { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'side_effect', arguments: '{}' } }, + ...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk => + 'index' in chunk ? { ...chunk, index: 1 } : chunk), + ] as Script[number] + const { ctx, parent } = await setup([response]) + let sideEffectRan = false + ctx.tools.register({ + name: 'side_effect', + description: 'probe', + parameters: { type: 'object', properties: {} }, + execute(): Promise { + sideEffectRan = true + return Promise.resolve([{ type: 'text', text: 'ran' }]) + }, + }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + // The call ran BEFORE captured was set: the deny gate only guards the + // window after the terminal answer landed. + expect(sideEffectRan).toBe(true) + expect(result.structured).toEqual({ answer: 6 }) + await run.dispose() + }) + + it('snapshots the schema at start(): caller mutation after start cannot drift enforcement', async () => { + const mutable: StructuredOutputSchema = { + type: 'object', + properties: { answer: { type: 'number' } }, + required: ['answer'], + additionalProperties: false, + } + const pristine = structuredClone(mutable) + const { ctx, parent, adapter } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }), + ]) + const run = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: mutable })) + // Mutate the caller's object AFTER start() returned but before the child's + // first request assembles: with a live reference this would reach both the + // model-visible parameters and validateStructuredValue. + ;(mutable.properties as Record).answer = { type: 'string' } + const result = await run.result + expect(result.structured).toEqual({ answer: 3 }) + // The child's request carried the PRISTINE schema, not the mutated one. + const childRequest = adapter.requests.at(-1) + const captureTool = (childRequest?.tools ?? []).find(tool => tool.name === STRUCTURED_OUTPUT_TOOL) + expect(captureTool?.parameters).toEqual(pristine) + await run.dispose() + }) + it('the captured-turn veto is prepend: an EARLIER force-continue listener cannot short-circuit it', async () => { const ctx = new Context() await ctx.plugin(SystemPrompt) From b907c20213e67805c1a2da097986683932c37114 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 09:14:48 +0800 Subject: [PATCH 03/24] review: drop the structured-output nudge; FIXME the context-global registry constraint MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two human review directives: - No re-prompt. A structured child that finishes a turn cleanly without calling structured_output settles error to the parent immediately — readResult already carried that mapping; the nudge loop only delayed it. Deletes the loop, its cancellation-window guard, STRUCTURED_OUTPUT_NUDGE, and the structuredNudgeRetries Config on both backends. - FIXME in the structured module doc: per-agent/per-session tool registry and prompt assembly would dissolve the final-assembly enforcement dance (the placeholder tool, the swap, the strip, the global-registration lifetime). --- packages/subagent/subagent-fork/README.md | 1 - packages/subagent/subagent-fork/src/index.ts | 17 +----- .../tests/multi-subagent.spec.ts | 4 +- .../subagent-fork/tests/subagent-fork.spec.ts | 4 +- .../subagent/subagent-inprocess/README.md | 4 +- .../subagent/subagent-inprocess/src/index.ts | 42 ++----------- .../subagent-inprocess/src/structured.ts | 12 ++-- .../tests/structured.spec.ts | 61 ++++++------------- .../tests/subagent-inprocess.spec.ts | 6 +- packages/subagent/subagent-spawn/README.md | 3 +- packages/subagent/subagent-spawn/src/index.ts | 25 ++------ .../subagent/subagent-spawn/tests/harness.ts | 2 +- .../tests/subagent-spawn.spec.ts | 4 +- 13 files changed, 50 insertions(+), 135 deletions(-) diff --git a/packages/subagent/subagent-fork/README.md b/packages/subagent/subagent-fork/README.md index 7b43f82261..1abd7951fd 100644 --- a/packages/subagent/subagent-fork/README.md +++ b/packages/subagent/subagent-fork/README.md @@ -19,6 +19,5 @@ The seam this rides on: `CreateAgentOptions.seed` (added on `dsh-agent`, threade | Key | Meaning | |---|---| | `providerName` | Registry name on `ctx.subagents` (default `fork`). | -| `structuredNudgeRetries` | How many times a structured run re-prompts a child that finished cleanly without calling `structured_output` (default 1). | See [`dsh-subagent-spawn`](../subagent-spawn/README.md) for the run lifecycle, model inheritance, and depth tracking — all shared. diff --git a/packages/subagent/subagent-fork/src/index.ts b/packages/subagent/subagent-fork/src/index.ts index 10d0492198..4ee28001d2 100644 --- a/packages/subagent/subagent-fork/src/index.ts +++ b/packages/subagent/subagent-fork/src/index.ts @@ -34,20 +34,14 @@ export const name = 'subagent-fork' // model-visible tool list) is unchanged by structured output. export const inject = ['subagents', 'agents'] -/** Config: the registry name to register the provider under, plus structured-run tuning. */ +/** Config: the registry name to register the provider under. */ export interface Config { /** Provider name on `ctx.subagents` (default `fork`). */ providerName: string - /** - * How many times a structured run re-prompts a child that finished cleanly - * without calling `structured_output` before giving up (default 1). - */ - structuredNudgeRetries: number } export const Config: z = z.object({ providerName: z.string().default('fork'), - structuredNudgeRetries: z.natural().default(1), }) /** @@ -76,17 +70,12 @@ class ForkProvider implements SubagentProvider { // Context contract: a forked child IS seeded with the parent's completed-turn prefix. readonly inheritsParentContext = true - constructor( - readonly name: string, - private readonly ctx: Context, - private readonly structuredNudgeRetries: number, - ) {} + constructor(readonly name: string, private readonly ctx: Context) {} start(request: SubagentStartRequest) { const seed = completedTurnPrefix(request.parent) return startInProcessRun(this.ctx, request, { providerName: this.name, - structuredNudgeRetries: this.structuredNudgeRetries, // Only pass a seed when there's a completed turn to inherit; an empty seed // is equivalent to a fresh child, so omit it to keep the session unseeded. ...seed.length > 0 ? { seed } : {}, @@ -102,5 +91,5 @@ export function apply(ctx: Context, config: Config): void { const acquisition = acquireStructuredRuntime(ctx) return () => { acquisition.release() } }, 'subagent-fork structured runtime') - ctx.subagents.registerProvider(new ForkProvider(config.providerName, ctx, config.structuredNudgeRetries)) + ctx.subagents.registerProvider(new ForkProvider(config.providerName, ctx)) } diff --git a/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts b/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts index 82caf25948..1f932fbaf9 100644 --- a/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts +++ b/packages/subagent/subagent-fork/tests/multi-subagent.spec.ts @@ -30,8 +30,8 @@ async function setup(script: Script) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - await ctx.plugin(Spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) - await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: 1 }) + await ctx.plugin(Spawn, { providerName: 'spawn' }) + await ctx.plugin(fork, { providerName: 'fork' }) ctx.llm.registerAdapter(['mock'], new MockAdapter(script)) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) return { ctx, parent } diff --git a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts index 2545188b52..74974942b5 100644 --- a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts +++ b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts @@ -37,7 +37,7 @@ async function setup(script: Script) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: 1 }) + await ctx.plugin(fork, { providerName: 'fork' }) ctx.llm.registerAdapter(['mock'], new MockAdapter(script)) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) return { ctx, parent } @@ -176,7 +176,7 @@ describe('dsh-subagent-fork', () => { // the registries are loaded here so the runtime registers eagerly anyway. await ctx.plugin(SystemPrompt) await ctx.plugin(ToolRegistry) - const fiber = await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: 1 }) + const fiber = await ctx.plugin(fork, { providerName: 'fork' }) expect(ctx.subagents.list()).toEqual(['fork']) await fiber.dispose() expect(ctx.subagents.list()).toEqual([]) diff --git a/packages/subagent/subagent-inprocess/README.md b/packages/subagent/subagent-inprocess/README.md index b816da8946..a4fcc51fbc 100644 --- a/packages/subagent/subagent-inprocess/README.md +++ b/packages/subagent/subagent-inprocess/README.md @@ -10,14 +10,14 @@ Runs a child as a child [`Agent`](../../core/agent) on the same cordis context ( 1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability); a `request.outputSchema` is asserted against the supported subset (`assertSupportedOutputSchema` from [dsh-tools](../../core/tools/README.md)) before any child exists; 2. creates a child via `ctx.agents.create` with a fresh `AgentId`/`SessionId`, the parent's `cwd` + `parentSession` lineage, the optional `options.seed` (fork's completed-turn prefix; omitted for a fresh child), and `agentOptions` (the child inherits the **parent's model** by default — a child with no model can't run — overridable via `request.agentOptions.model`; the deployment persona needs no inheritance — it is a context-wide prompt section); -3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts); a structured child that finished a turn CLEANLY without calling `structured_output` is re-prompted (a nudge — a fresh turn) up to `options.structuredNudgeRetries` times; +3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts); there is deliberately NO re-prompt for a structured child that finished cleanly without calling `structured_output` — the shortfall maps to an `error` result for the parent; 4. reads the result, scoped to the child's OWN events (everything at or after `seedLength`, so a seeded child that produced no message of its own never returns the seeded parent's last message): the last `assistant/message` content (deep-cloned — the log is frozen) and the last `turn/end.reason` mapped to a `SubagentStopReason`. A structured run surfaces the captured value as `result.structured`; a structured child that finished cleanly WITHOUT ever capturing settles `error` (a clean finish without the demanded result is a failure, not a success with a missing field). `dispose()` delegates to `AgentHandle.dispose()` (stop loop → await quiescence → remove session); `cancel()` cancels the child's in-flight turn. A cancel landing before any `turn/end` (the pre-turn window) still settles `aborted`, honoring the cancel contract rather than the generic no-turn `error`. ### `InProcessRunOptions` -`{ providerName: string; seed?: SessionEvent[]; structuredNudgeRetries: number }` — the per-backend inputs: the provider name (for error context), the optional child-session seed, and the structured-run nudge budget (REQUIRED, resolved from the backend's validated Config — the driver never fills it with a hidden default). +`{ providerName: string; seed?: SessionEvent[] }` — the per-backend inputs: the provider name (for error context) and the optional child-session seed. ### Structured output: `acquireStructuredRuntime(ctx): StructuredAcquisition` diff --git a/packages/subagent/subagent-inprocess/src/index.ts b/packages/subagent/subagent-inprocess/src/index.ts index 1309556f3f..72906a78fc 100644 --- a/packages/subagent/subagent-inprocess/src/index.ts +++ b/packages/subagent/subagent-inprocess/src/index.ts @@ -22,7 +22,6 @@ import { assertSupportedOutputSchema } from '@deepseek-ai/dsh-tools' import type { SubagentResult, SubagentRun, SubagentStartRequest, SubagentStopReason } from '@deepseek-ai/dsh-subagent' import { acquireStructuredRuntime, - STRUCTURED_OUTPUT_NUDGE, type StructuredAcquisition, } from './structured.ts' @@ -30,7 +29,6 @@ export { acquireStructuredRuntime, STRUCTURED_OUTPUT_TOOL, STRUCTURED_OUTPUT_INSTRUCTION, - STRUCTURED_OUTPUT_NUDGE, type StructuredAcquisition, } from './structured.ts' @@ -90,13 +88,6 @@ export interface InProcessRunOptions { * parent's log (FORK), or `undefined` for a fresh child (SPAWN). */ readonly seed?: SessionEvent[] - /** - * How many times a structured run re-prompts a child that finished a turn - * cleanly WITHOUT calling `structured_output` (see the structured module). - * REQUIRED, resolved from the backend's validated Config — per the explicit- - * defaulting rule, the driver never fills it with a hidden fallback. - */ - readonly structuredNudgeRetries: number } /** @@ -178,7 +169,7 @@ export function startInProcessRun( let cancelled = false // An accessor, not an inline read: `cancelled` mutates from closures (the // abort listener, run.cancel), which control-flow narrowing cannot see — an - // inline `!cancelled` in the nudge condition reads as always-true. + // inline read at the result mapping would narrow to the initializer. const isCancelled = (): boolean => cancelled const requestCancel = (reason: string): void => { cancelled = true @@ -196,28 +187,9 @@ export function startInProcessRun( if (request.signal?.aborted) return { output: [], stopReason: 'aborted' } child.send(request.prompt) await child.whenIdle() - if (structured) { - // Nudge loop: a child that finished a turn CLEANLY without calling - // structured_output gets re-prompted, up to the backend-configured - // retry count. An errored/aborted turn is not nudged — its failure is - // the honest result (a cancelled turn ends `aborted`, and a pre-turn - // cancel leaves no `turn/end` at all, so neither reads `completed`). - // `!cancelled` closes the remaining window: a cancel landing AFTER a - // clean turn end clears nothing — `child.cancel()` only kills - // queued/running work — so without it the next `send` would spend a - // fresh post-cancellation turn; the condition re-evaluates after - // every `whenIdle()`, so a mid-nudge cancel stops the loop at the - // next boundary too. - let nudges = options.structuredNudgeRetries - while ( - !isCancelled() && structured.captured(child) === undefined && nudges > 0 - && lastOwnTurnEnd(child, seedLength)?.data.reason.kind === 'completed' - ) { - nudges -= 1 - child.send([{ type: 'text', text: STRUCTURED_OUTPUT_NUDGE }]) - await child.whenIdle() - } - } + // Deliberately NO re-prompt when a structured child finishes cleanly + // without calling structured_output: readResult maps that to `error` — + // the shortfall goes to the parent instead of buying extra model turns. return readResult(child, seedLength, isCancelled(), structured ? { captured: structured.captured(child) } : undefined) } finally { request.signal?.removeEventListener('abort', onAbort) @@ -241,12 +213,6 @@ export function startInProcessRun( } } -/** The child's OWN last `turn/end` event (events at or after `seedLength`), if any. */ -function lastOwnTurnEnd(child: Agent, seedLength: number): SessionEvent<'turn/end'> | undefined { - return child.session.events.slice(seedLength) - .findLast((e): e is SessionEvent<'turn/end'> => e.type === 'turn/end') -} - /** * Read a settled child's terminal result from its session log, scoped to the * child's OWN events (everything at or after `seedLength` — fork seeds the diff --git a/packages/subagent/subagent-inprocess/src/structured.ts b/packages/subagent/subagent-inprocess/src/structured.ts index e866e5e699..370f7a538d 100644 --- a/packages/subagent/subagent-inprocess/src/structured.ts +++ b/packages/subagent/subagent-inprocess/src/structured.ts @@ -21,6 +21,13 @@ * returning a replacement assembly — see the waterfall composition caveat in * docs/architecture.md.) * + * FIXME: the whole enforcement dance above exists because the tool registry + * and prompt assembly are context-global. If they become per-agent or + * per-session scoped, a structured run just registers its own schema'd tool on + * the child's scope and this module reduces to the capture tool plus the + * turn-stop — no placeholder, no final-assembly swap, no strip-for-everyone- + * else, no global-registration lifetime dance. + * * A companion `agent/turn-continuation` listener stops a child's turn once its * output is captured — without it, the loop's default "had tool calls ⇒ * continue" buys a wasted extra model step per structured child. It is also @@ -65,11 +72,6 @@ export const STRUCTURED_OUTPUT_INSTRUCTION + `\`${STRUCTURED_OUTPUT_TOOL}\` tool with arguments matching its parameter schema exactly. ` + 'Do not finish with a plain text answer: only the tool call counts as your result.' -/** The nudge sent when a structured child finishes cleanly without calling the tool. */ -export const STRUCTURED_OUTPUT_NUDGE - = `You finished without calling \`${STRUCTURED_OUTPUT_TOOL}\`. ` - + `Call \`${STRUCTURED_OUTPUT_TOOL}\` now with your final result matching its parameter schema.` - /** One structured run's state: the schema to enforce and the captured value, once recorded. */ interface RunState { readonly schema: StructuredOutputSchema diff --git a/packages/subagent/subagent-inprocess/tests/structured.spec.ts b/packages/subagent/subagent-inprocess/tests/structured.spec.ts index 006fcd9d8a..ba693cfecd 100644 --- a/packages/subagent/subagent-inprocess/tests/structured.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/structured.spec.ts @@ -32,7 +32,7 @@ const SCHEMA: StructuredOutputSchema = { * structured runtime at apply, exactly as shipped). The mock model script * drives the child's structured_output calls. */ -async function setup(script: Script, options?: { nudges?: number; withFork?: boolean }) { +async function setup(script: Script, options?: { withFork?: boolean }) { const ctx = new Context() const adapter = new MockAdapter(script) await ctx.plugin(LlmService) @@ -43,9 +43,9 @@ async function setup(script: Script, options?: { nudges?: number; withFork?: boo await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - const fiber = await ctx.plugin(spawn, { providerName: 'spawn', structuredNudgeRetries: options?.nudges ?? 1 }) + const fiber = await ctx.plugin(spawn, { providerName: 'spawn' }) const forkFiber = options?.withFork - ? await ctx.plugin(fork, { providerName: 'fork', structuredNudgeRetries: options?.nudges ?? 1 }) + ? await ctx.plugin(fork, { providerName: 'fork' }) : undefined ctx.llm.registerAdapter(['mock'], adapter) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) @@ -215,47 +215,25 @@ describe('in-process structured output', () => { await run.dispose() }) - it('nudges a child that finished cleanly without calling the tool, then captures', async () => { - const { ctx, parent } = await setup([ - textResponse('here is my answer in prose'), - toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }), - ]) - const run = ctx.subagents.start('spawn', structuredRequest(parent)) - const result = await run.result - expect(result.structured).toEqual({ answer: 3 }) - expect(result.stopReason).toBe('completed') - // The nudge is a real user-visible message in the child's log. - const child = ctx.agents.get(run.id)! - const users = child.session.events.filter(e => e.type === 'user/message') - expect(users.length).toBe(2) - await run.dispose() - }) - - it('settles error when the nudges run out without a capture', async () => { + it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => { const { ctx, parent, adapter } = await setup([ - textResponse('prose only'), - textResponse('still prose'), - ], { nudges: 1 }) + textResponse('here is my answer in prose'), + textResponse('MUST NOT BE CONSUMED'), + ]) const run = ctx.subagents.start('spawn', structuredRequest(parent)) const result = await run.result expect(result.stopReason).toBe('error') expect(result.structured).toBeUndefined() - expect(adapter.requests.length).toBe(2) - await run.dispose() - }) - - it('zero nudge retries fails immediately after the first clean prose finish', async () => { - const { ctx, parent, adapter } = await setup([textResponse('prose')], { nudges: 0 }) - const run = ctx.subagents.start('spawn', structuredRequest(parent)) - const result = await run.result - expect(result.stopReason).toBe('error') + // Exactly one model request and one user message: no nudge turn exists. expect(adapter.requests.length).toBe(1) + const child = ctx.agents.get(run.id)! + expect(child.session.events.filter(e => e.type === 'user/message').length).toBe(1) await run.dispose() }) - it('a child that errored is NOT nudged (its failure is the honest result)', async () => { + it('an errored child keeps its honest error result (no capture expected)', async () => { // Script exhaustion on the first call → the child turn errors. - const { ctx, parent, adapter } = await setup([], { nudges: 3 }) + const { ctx, parent, adapter } = await setup([]) const run = ctx.subagents.start('spawn', structuredRequest(parent)) const result = await run.result expect(result.stopReason).toBe('error') @@ -263,22 +241,17 @@ describe('in-process structured output', () => { await run.dispose() }) - it('a cancel landing after a clean turn end stops the nudge loop: no post-cancellation turn is spent', async () => { - const { ctx, parent, adapter } = await setup([textResponse('prose, no capture')], { nudges: 3 }) + it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => { + const { ctx, parent } = await setup([textResponse('prose, no capture')]) const run = ctx.subagents.start('spawn', structuredRequest(parent)) const child = ctx.agents.get(run.id)! - // Cancel synchronously inside the first turn's end recording — after the - // turn reads `completed`, before the nudge continuation resumes. The turn - // state alone cannot see this cancel (`child.cancel()` only clears - // queued/running work), so without the loop's own cancelled check the - // next send would spend a fresh child turn after the caller cancelled. + // Cancel synchronously inside the turn's end recording: the cancel + // contract outranks the schema shortfall, so the result maps to aborted. ctx.on('session/event', (session, event) => { - if (session === child.session && event.type === 'turn/end') run.cancel('cancelled between turn end and nudge') + if (session === child.session && event.type === 'turn/end') run.cancel('cancelled at turn end') }) const result = await run.result expect(result.stopReason).toBe('aborted') - // Exactly one model request: the nudge turn never ran. - expect(adapter.requests.length).toBe(1) await run.dispose() }) diff --git a/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts b/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts index d3870ae51a..7219e03988 100644 --- a/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/subagent-inprocess.spec.ts @@ -51,7 +51,7 @@ describe('depthOf', () => { describe('startInProcessRun', () => { it('drives a fresh child (no seed) to completion and returns its output', async () => { const { ctx, parent } = await setup([textResponse('driver child answer')]) - const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'do X' }], parent }, { providerName: 'spawn', structuredNudgeRetries: 1 }) + const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'do X' }], parent }, { providerName: 'spawn' }) const result = await run.result expect(result.stopReason).toBe('completed') expect(text(result.output)).toBe('driver child answer') @@ -61,7 +61,7 @@ describe('startInProcessRun', () => { it('throws SubagentDepthError when the child would exceed maxDepth', async () => { const { ctx, parent } = await setup([]) - expect(() => startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'p' }], parent, maxDepth: 0 }, { providerName: 'spawn', structuredNudgeRetries: 1 })) + expect(() => startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'p' }], parent, maxDepth: 0 }, { providerName: 'spawn' })) .toThrow(SubagentDepthError) }) @@ -73,7 +73,7 @@ describe('startInProcessRun', () => { parent.send([{ type: 'text', text: 'parent q' }]) await parent.whenIdle() const seed = parent.session.events.slice() - const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'child q' }], parent }, { providerName: 'fork', structuredNudgeRetries: 1, seed }) + const run = startInProcessRun(ctx, { prompt: [{ type: 'text', text: 'child q' }], parent }, { providerName: 'fork', seed }) const result = await run.result expect(result.stopReason).toBe('completed') expect(text(result.output)).toBe('seeded child reply') diff --git a/packages/subagent/subagent-spawn/README.md b/packages/subagent/subagent-spawn/README.md index 059c996215..696007a693 100644 --- a/packages/subagent/subagent-spawn/README.md +++ b/packages/subagent/subagent-spawn/README.md @@ -6,7 +6,7 @@ The run mechanics live in the shared [`@deepseek-ai/dsh-subagent-inprocess`](../ ## What it does -`start(request)` delegates to `startInProcessRun(ctx, request, { providerName, structuredNudgeRetries })` with no seed: a fresh child agent with the parent's `cwd`/`parentSession` lineage and (by default) the parent's model. See the [driver README](../subagent-inprocess/README.md) for the full lifecycle (depth check, one-shot drive, result read, dispose). +`start(request)` delegates to `startInProcessRun(ctx, request, { providerName })` with no seed: a fresh child agent with the parent's `cwd`/`parentSession` lineage and (by default) the parent's model. See the [driver README](../subagent-inprocess/README.md) for the full lifecycle (depth check, one-shot drive, result read, dispose). ## Capabilities @@ -17,4 +17,3 @@ The run mechanics live in the shared [`@deepseek-ai/dsh-subagent-inprocess`](../ | Key | Meaning | |---|---| | `providerName` | Registry name on `ctx.subagents` (default `spawn`). | -| `structuredNudgeRetries` | How many times a structured run re-prompts a child that finished cleanly without calling `structured_output` (default 1). | diff --git a/packages/subagent/subagent-spawn/src/index.ts b/packages/subagent/subagent-spawn/src/index.ts index 248e887ce7..7da954bc44 100644 --- a/packages/subagent/subagent-spawn/src/index.ts +++ b/packages/subagent/subagent-spawn/src/index.ts @@ -32,20 +32,14 @@ export const name = 'subagent-spawn' // structured output existed. export const inject = ['subagents', 'agents'] -/** Config: the registry name to register the provider under, plus structured-run tuning. */ +/** Config: the registry name to register the provider under. */ export interface Config { /** Provider name on `ctx.subagents` (default `spawn`). */ providerName: string - /** - * How many times a structured run re-prompts a child that finished cleanly - * without calling `structured_output` before giving up (default 1). - */ - structuredNudgeRetries: number } export const Config: z = z.object({ providerName: z.string().default('spawn'), - structuredNudgeRetries: z.natural().default(1), }) /** @@ -59,20 +53,13 @@ class SpawnProvider implements SubagentProvider { // Context contract: a spawned child starts fresh — it never sees the parent conversation. readonly inheritsParentContext = false - constructor( - readonly name: string, - private readonly ctx: Context, - private readonly structuredNudgeRetries: number, - ) {} + constructor(readonly name: string, private readonly ctx: Context) {} start(request: SubagentStartRequest) { // Fresh child: no seed. The shared driver mints ids, stamps cwd/lineage/ - // depth, drives the one-shot (including the structured capture/nudge loop - // when the request carries an outputSchema), and maps the result. - return startInProcessRun(this.ctx, request, { - providerName: this.name, - structuredNudgeRetries: this.structuredNudgeRetries, - }) + // depth, drives the one-shot (including the structured capture when the + // request carries an outputSchema), and maps the result. + return startInProcessRun(this.ctx, request, { providerName: this.name }) } } @@ -85,5 +72,5 @@ export function apply(ctx: Context, config: Config): void { const acquisition = acquireStructuredRuntime(ctx) return () => { acquisition.release() } }, 'subagent-spawn structured runtime') - ctx.subagents.registerProvider(new SpawnProvider(config.providerName, ctx, config.structuredNudgeRetries)) + ctx.subagents.registerProvider(new SpawnProvider(config.providerName, ctx)) } diff --git a/packages/subagent/subagent-spawn/tests/harness.ts b/packages/subagent/subagent-spawn/tests/harness.ts index 97dab3c3ee..b3e9d4ec24 100644 --- a/packages/subagent/subagent-spawn/tests/harness.ts +++ b/packages/subagent/subagent-spawn/tests/harness.ts @@ -34,7 +34,7 @@ export async function spawnHarness(workdir: string): Promise { await ctx.plugin(LocalBashExecutor, { cwd: workdir, timeoutMs: 30_000 }) await ctx.plugin(ToolBash) await ctx.plugin(SubagentService) - await ctx.plugin(Spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) + await ctx.plugin(Spawn, { providerName: 'spawn' }) // The model-facing subagent tool, bound to the spawn backend. await ctx.plugin(ToolSubagent, { provider: 'spawn' }) return ctx diff --git a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts index 70ce7a4774..63d26c0531 100644 --- a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts +++ b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts @@ -34,7 +34,7 @@ async function setup(script: Script) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - await ctx.plugin(spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) + await ctx.plugin(spawn, { providerName: 'spawn' }) ctx.llm.registerAdapter(['mock'], adapter) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) return { ctx, parent, adapter } @@ -257,7 +257,7 @@ describe('dsh-subagent-spawn', () => { // the registries are loaded here so the runtime registers eagerly anyway. await ctx.plugin(SystemPrompt) await ctx.plugin(ToolRegistry) - const fiber = await ctx.plugin(spawn, { providerName: 'spawn', structuredNudgeRetries: 1 }) + const fiber = await ctx.plugin(spawn, { providerName: 'spawn' }) expect(ctx.subagents.list()).toEqual(['spawn']) await fiber.dispose() expect(ctx.subagents.list()).toEqual([]) From 5a91893b860b2b80f8bf9a7d5a71da85b5f53f89 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 09:37:50 +0800 Subject: [PATCH 04/24] test: drain the detached SubagentStart continuation before the bridge spec ends MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The markers are touched mid-script, so runPoint's continuation chain (child exit → merge → the listener's detached .then) can still be in flight when the marker poll resolves; a vitest worker that exits first leaves the inject condition's short-circuit path uncounted. Observed as a CI-only 99.03% branch-coverage flake on hooks-claude — surfaced by this branch shifting suite timing, latent before it. Two macrotask rounds pin the path deterministically. --- packages/hooks/hooks-claude/tests/bridge.spec.ts | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/packages/hooks/hooks-claude/tests/bridge.spec.ts b/packages/hooks/hooks-claude/tests/bridge.spec.ts index 3e36231e66..652edf66b7 100644 --- a/packages/hooks/hooks-claude/tests/bridge.spec.ts +++ b/packages/hooks/hooks-claude/tests/bridge.spec.ts @@ -298,6 +298,14 @@ describe('hooks-claude bridge — SubagentStart / SubagentStop (observe)', () => await waitFor(() => existsSync(startMarker) && existsSync(stopMarker)) expect(existsSync(startMarker)).toBe(true) expect(existsSync(stopMarker)).toBe(true) + // The markers are touched MID-script, so runPoint's continuation chain + // (child-exit event → merge → the listener's detached .then) can still be + // in flight when the poll resolves. Drain two macrotask rounds so the + // short-circuit path of the SubagentStart inject condition executes before + // this worker can exit — observed as a CI-only 99.03% branch-coverage + // flake on hooks-claude when the worker won the race. + await new Promise(resolve => setTimeout(resolve, 0)) + await new Promise(resolve => setTimeout(resolve, 0)) }) }) From 8c8189844f6bf35c697132ff593b2aeb176b9439 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 09:44:18 +0800 Subject: [PATCH 05/24] docs: regenerate the config catalog after the master merge Master's generated config catalog (#188, flattened paths #191) now records plugin Configs; the nudge removal dropped structuredNudgeRetries from both backends, so the regenerated catalog loses those rows. --- docs/config-catalog.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 66766144a8..8a10012050 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -218,7 +218,7 @@ export interface Config { } ``` -Source: [`packages/hooks/hooks-claude/src/index.ts:55`](../packages/hooks/hooks-claude/src/index.ts) +Source: [`packages/hooks/hooks-claude/src/index.ts:56`](../packages/hooks/hooks-claude/src/index.ts) ## `@deepseek-ai/dsh-hooks-codex` @@ -243,7 +243,7 @@ export interface Config { } ``` -Source: [`packages/hooks/hooks-codex/src/index.ts:42`](../packages/hooks/hooks-codex/src/index.ts) +Source: [`packages/hooks/hooks-codex/src/index.ts:43`](../packages/hooks/hooks-codex/src/index.ts) ## `@deepseek-ai/dsh-invariants` @@ -329,7 +329,7 @@ export interface Config { } ``` -Source: [`packages/support/llm-replay/src/index.ts:411`](../packages/support/llm-replay/src/index.ts) +Source: [`packages/support/llm-replay/src/index.ts:415`](../packages/support/llm-replay/src/index.ts) ## `@deepseek-ai/dsh-session-persistence-jsonl` @@ -481,7 +481,7 @@ export interface Config { } ``` -Source: [`packages/subagent/subagent-fork/src/index.ts:34`](../packages/subagent/subagent-fork/src/index.ts) +Source: [`packages/subagent/subagent-fork/src/index.ts:38`](../packages/subagent/subagent-fork/src/index.ts) ## `@deepseek-ai/dsh-subagent-mock` @@ -528,7 +528,7 @@ export interface Config { } ``` -Source: [`packages/subagent/subagent-spawn/src/index.ts:26`](../packages/subagent/subagent-spawn/src/index.ts) +Source: [`packages/subagent/subagent-spawn/src/index.ts:36`](../packages/subagent/subagent-spawn/src/index.ts) ## `@deepseek-ai/dsh-system-prompt` From f33e14ff19721f372cef9ceefdfc858ebc757240 Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Mon, 6 Jul 2026 11:46:10 +0800 Subject: [PATCH 06/24] build: lower the Node engines floor to 22.18 --- .github/workflows/ci.yml | 4 +- .github/workflows/e2e.yml | 13 +++++- AGENTS.md | 2 +- docs/core-data-structures/web.md | 2 +- docs/development.i18n.yaml | 4 +- docs/development.md | 4 +- docs/development.zh.md | 4 +- docs/rfc/INDEX.md | 1 + .../2026-06-24-web-capability-seam.md | 2 +- .../process/2026-06-11-quality-gates.md | 2 +- .../process/2026-07-06-node-22-18-floor.md | 30 +++++++++++++ .../testing/2026-06-19-real-api-e2e-ci.md | 2 +- package.json | 4 +- .../session-persistence-sqlite/README.md | 2 +- .../ui/stdio-agent/tests/built-bin.e2e.ts | 7 ++-- packages/web/web-fetch-local/src/provider.ts | 2 +- .../web/web-search-deepseek/src/provider.ts | 2 +- packages/web/web-search-exa/src/provider.ts | 2 +- .../web/web-search-perplexity/src/provider.ts | 2 +- pnpm-lock.yaml | 42 ++++++++++++------- 20 files changed, 94 insertions(+), 39 deletions(-) create mode 100644 docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index efa4dfc458..52d1ce75fa 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -95,13 +95,13 @@ jobs: node-compat: runs-on: ubuntu-latest - name: node 26 + name: node ${{ matrix.node }} env: DSH_GATE_CONCURRENCY: '2' strategy: fail-fast: false matrix: - node: [26] + node: ['22.18', 24, 26] steps: - uses: actions/checkout@v6 diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 1ae0733286..eb73308b50 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -49,6 +49,17 @@ permissions: jobs: e2e: runs-on: ubuntu-latest + strategy: + fail-fast: false + matrix: + # The keyless ci.yml matrix runs the MOCK adapter (dsh-llm-replay); the + # real fetch + SSE-streaming adapter path runs ONLY here, so its + # node-version compat is covered nowhere else. Run the real-API suite on + # the engines floor AND the primary line to close that gap. 26 is left to + # the keyless matrix — floor + LTS is the meaningful pair for the live + # network path, and inference is cheap (we are DeepSeek). + node: ['22.18', 24] + name: e2e node ${{ matrix.node }} # Run on every trusted event. Skip untrusted PRs (forks + Dependabot) where # the secret is withheld — they would otherwise hard-fail the preflight. if: >- @@ -62,7 +73,7 @@ jobs: - uses: actions/setup-node@v6 with: - node-version: 24 + node-version: ${{ matrix.node }} - name: Enable corepack (pnpm) run: corepack enable diff --git a/AGENTS.md b/AGENTS.md index c582053122..c97ca15b55 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -34,7 +34,7 @@ Per-package map: the group READMEs, indexed from [packages/README.md](packages/R ## Commands ```sh -pnpm install # pnpm workspaces, node >= 24 +pnpm install # pnpm workspaces, node >= 22.18 pnpm run test # vitest unit tests pnpm run test:coverage # THE gating test run: per-file 100% coverage on packages/*/*/src pnpm run test:e2e # real-API tests; self-skip without DEEPSEEK_API_KEY diff --git a/docs/core-data-structures/web.md b/docs/core-data-structures/web.md index 05c3f648c0..00f10278f2 100644 --- a/docs/core-data-structures/web.md +++ b/docs/core-data-structures/web.md @@ -91,4 +91,4 @@ Selection never depends on registration, config, or HMR order: a capability has ## The service -`WebService` (`ctx.web`, defined in [`packages/web/web/src/index.ts`](../../packages/web/web/src/index.ts)) is a provider registry plus a provider-selecting execution surface, close to `LlmService`'s shape: `registerSearchProvider`/`registerFetchProvider` (duplicate ids throw `WEB_DUPLICATE_PROVIDER`, return disposers) and `search`/`fetch` (resolve the provider at call time, throw a structured `WebError` when the capability cannot run). Providers issue requests with the platform-native `fetch` (Node 24), mirroring `dsh-llm-deepseek`; the `dsh-web-fetch-local` provider owns safe retrieval (http/https-only, credential rejection, byte/char/timeout/redirect caps, same-origin-only redirects with per-hop re-validation, charset decoding) while `dsh-tool-web` owns presentation (HTML→markdown). SSRF / private-network blocking is deferred (see the RFC) — until it lands, `web_fetch` must not be enabled where it can reach sensitive internal targets. +`WebService` (`ctx.web`, defined in [`packages/web/web/src/index.ts`](../../packages/web/web/src/index.ts)) is a provider registry plus a provider-selecting execution surface, close to `LlmService`'s shape: `registerSearchProvider`/`registerFetchProvider` (duplicate ids throw `WEB_DUPLICATE_PROVIDER`, return disposers) and `search`/`fetch` (resolve the provider at call time, throw a structured `WebError` when the capability cannot run). Providers issue requests with the platform-native `fetch` (Node 22.18), mirroring `dsh-llm-deepseek`; the `dsh-web-fetch-local` provider owns safe retrieval (http/https-only, credential rejection, byte/char/timeout/redirect caps, same-origin-only redirects with per-hop re-validation, charset decoding) while `dsh-tool-web` owns presentation (HTML→markdown). SSRF / private-network blocking is deferred (see the RFC) — until it lands, `web_fetch` must not be enabled where it can reach sensitive internal targets. diff --git a/docs/development.i18n.yaml b/docs/development.i18n.yaml index 2ab3478a61..2fc2c3cfe9 100644 --- a/docs/development.i18n.yaml +++ b/docs/development.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -development.md: 97ca3f6b9fc9653ab658e480e6155fc1e121854f -development.zh.md: e837afb6a01ed4d0c4801886bd6ca6a7602ac573 +development.md: 8f901909264d4405d396782d68c28fbde9b85bfa +development.zh.md: 2b2af080d11b8ea9d9cbf26c62833b4fc00fb6ba diff --git a/docs/development.md b/docs/development.md index 97ca3f6b9f..8f90190926 100644 --- a/docs/development.md +++ b/docs/development.md @@ -6,7 +6,7 @@ This guide covers the local setup needed to work on DeepSeek Harness and underst ## Prerequisites -- Node.js 24 or newer. The repo declares `node >=24`; CI runs the matrix on Node 24 and 26. +- Node.js 22.18 or newer. The repo declares `node >=22.18`; CI runs the matrix on Node 22.18, 24, and 26. - Corepack-enabled pnpm. The repo pins `pnpm@11.7.0` in `package.json`; run `corepack enable` if `pnpm --version` does not resolve through Corepack. - Git. - Optional: a DeepSeek API key for the REPL/ACP agent demos and real-API e2e tests. @@ -63,7 +63,7 @@ lefthook is configured in `lefthook.yml` as an early local checkpoint before rev The vendor manifest guard checks that changes under `vendor/*/src` are staged with the matching `vendor/README.md` manifest update. See `vendor/README.md` before editing vendored code. -These hooks do not exactly mirror CI. Notably, `pre-push` runs unit tests without coverage, while CI runs `pnpm run test:coverage`; CI also runs echo-agent and built-bin smoke tests and exercises the matrix on Node 24 and 26. +These hooks do not exactly mirror CI. Notably, `pre-push` runs unit tests without coverage, while CI runs `pnpm run test:coverage`; CI also runs echo-agent and built-bin smoke tests and exercises the matrix on Node 22.18, 24, and 26. ## CI gates diff --git a/docs/development.zh.md b/docs/development.zh.md index e837afb6a0..2b2af080d1 100644 --- a/docs/development.zh.md +++ b/docs/development.zh.md @@ -6,7 +6,7 @@ ## 前置条件 -- Node.js 24 或更新版本。仓库声明 `node >=24`;CI 在 Node 24 和 26 上跑矩阵。 +- Node.js 22.18 或更新版本。仓库声明 `node >=22.18`;CI 在 Node 22.18、24 和 26 上跑矩阵。 - 启用了 Corepack 的 pnpm。仓库在 `package.json` 中钉住 `pnpm@11.7.0`;如果 `pnpm --version` 无法通过 Corepack 解析,先运行 `corepack enable`。 - Git。 - 可选:一个 DeepSeek API key,用于 REPL/ACP agent(智能体)演示和真实 API 的 e2e 测试。 @@ -63,7 +63,7 @@ lefthook 在 `lefthook.yml` 中配置,作为评审前的本地早期检查点 vendor manifest 守卫检查 `vendor/*/src` 下的改动是否连同对应的 `vendor/README.md` manifest 更新一起暂存。编辑 vendor 代码前先看 `vendor/README.md`。 -这些钩子并不与 CI 完全一致。特别是:`pre-push` 跑不带覆盖率的单元测试,而 CI 跑 `pnpm run test:coverage`;CI 还会跑 echo-agent 和 built-bin 冒烟测试,并在 Node 24 和 26 上跑矩阵。 +这些钩子并不与 CI 完全一致。特别是:`pre-push` 跑不带覆盖率的单元测试,而 CI 跑 `pnpm run test:coverage`;CI 还会跑 echo-agent 和 built-bin 冒烟测试,并在 Node 22.18、24 和 26 上跑矩阵。 ## CI 门禁 diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index 5e23a876fb..6b15512503 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -144,6 +144,7 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; | [Generated persistence log event catalog](implemented/process/2026-07-04-persistence-log-catalog.md) | 2026-07-04 | | [One gated in-file format for RFCs](implemented/process/2026-07-05-uniform-rfc-format.md) | 2026-07-05 | | [Generated plugin config catalog](implemented/process/2026-07-06-generated-config-catalog.md) | 2026-07-06 | +| [Lower the Node engines floor to 22.18](implemented/process/2026-07-06-node-22-18-floor.md) | 2026-07-06 | | [Parallel GitHub CI gates](implemented/process/2026-07-06-parallel-github-ci-gates.md) | 2026-07-06 | | [Parallel pre-push gates](implemented/process/2026-07-06-parallel-pre-push-gates.md) | 2026-07-06 | diff --git a/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md b/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md index a03df16c8c..74b68ca2f3 100644 --- a/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md +++ b/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md @@ -66,7 +66,7 @@ flowchart LR `@deepseek-ai/dsh-web` depends only on Cordis and low-level harness support. It declares `ctx.web`, provider interfaces, request/result types, the provider status type, and error codes. It does not import tool, agent, session, LLM, or provider packages. -Provider packages depend on `@deepseek-ai/dsh-web` and Cordis. They own credentials, endpoint config, provider-specific request mapping, provider-specific response parsing, and provider-specific error translation into `WebError`. They issue network requests with the platform-native `fetch` (Node 24), mirroring `@deepseek-ai/dsh-llm-deepseek`'s adapter, NOT a cordis HTTP-client service (`ctx.http`/`@cordisjs/plugin-http`) — even where a Perplexity provider's request is shaped like an OpenAI-compatible chat completion, that wire shape is a provider-private detail and does not make the provider depend on `ctx.llm`. A provider does NOT own the `ctx.web` key (two search providers cannot both own it): like `dsh-llm-deepseek`, each provider package is a function/namespace plugin (`inject: ['web']`) whose `apply` constructs the backend and calls `ctx.web.registerSearchProvider` / `registerFetchProvider`. `@deepseek-ai/dsh-web` is the `export default` service that owns the key. +Provider packages depend on `@deepseek-ai/dsh-web` and Cordis. They own credentials, endpoint config, provider-specific request mapping, provider-specific response parsing, and provider-specific error translation into `WebError`. They issue network requests with the platform-native `fetch` (Node 22.18), mirroring `@deepseek-ai/dsh-llm-deepseek`'s adapter, NOT a cordis HTTP-client service (`ctx.http`/`@cordisjs/plugin-http`) — even where a Perplexity provider's request is shaped like an OpenAI-compatible chat completion, that wire shape is a provider-private detail and does not make the provider depend on `ctx.llm`. A provider does NOT own the `ctx.web` key (two search providers cannot both own it): like `dsh-llm-deepseek`, each provider package is a function/namespace plugin (`inject: ['web']`) whose `apply` constructs the backend and calls `ctx.web.registerSearchProvider` / `registerFetchProvider`. `@deepseek-ai/dsh-web` is the `export default` service that owns the key. `@deepseek-ai/dsh-tool-web` depends on `@deepseek-ai/dsh-web`, `@deepseek-ai/dsh-tools`, `@deepseek-ai/dsh-system-prompt`, and Cordis. It never imports concrete provider packages. diff --git a/docs/rfc/implemented/process/2026-06-11-quality-gates.md b/docs/rfc/implemented/process/2026-06-11-quality-gates.md index 69b1beb554..52823b8222 100644 --- a/docs/rfc/implemented/process/2026-06-11-quality-gates.md +++ b/docs/rfc/implemented/process/2026-06-11-quality-gates.md @@ -14,7 +14,7 @@ Every AGENTS.md promise gets a command that exits non-zero, wired into git hooks - ESLint strict-type-checked + @stylistic (the house style, enforced); vendored code excluded. - Per-file 100% coverage on `packages/*/src` (v8); unreachable defensive guards carry `/* v8 ignore */ ` with stated reasons instead of deletion. - knip (dead code/deps), publint (package correctness), workspace constraints (workspace rules: private, cordis peer+dev, uniform version, ESM), and a NodeNext consumer typecheck for built package declarations. -- lefthook pre-commit (lint staged, typecheck, vendor-manifest guard) and pre-push (tests, hygiene); CI runs the full matrix on node 24/26 plus a demo smoke test driving the echo-agent end to end. +- lefthook pre-commit (lint staged, typecheck, vendor-manifest guard) and pre-push (tests, hygiene); CI runs the full matrix on node 22.18/24/26 plus a demo smoke test driving the echo-agent end to end. ## Consequences diff --git a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md new file mode 100644 index 0000000000..737bb2e40f --- /dev/null +++ b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md @@ -0,0 +1,30 @@ +# RFC: Lower the Node engines floor to 22.18 + +Status: implemented + +## Problem + +The root `engines.node` was `>=24`, which excluded the entire Node 22 LTS line for no runtime reason. The harness has exactly two Node features whose availability gates the floor, and both are satisfied well below Node 24 — so the floor was higher than the code actually requires. Pinning it honestly widens the supported install base (Node 22 LTS is in service until 2027) without weakening any guarantee, provided CI proves the claim on the floor version rather than merely asserting it in a manifest. + +## Decision + +Set `engines.node` to `>=22.18` and treat 22.18 as the tested floor everywhere (CI matrix `['22.18', 24, 26]`, the real-API e2e job on `['22.18', 24]` — floor plus primary line, since the keyless matrix exercises only the mock adapter and the live `fetch`/SSE path runs only in e2e). 22.18 is the *later* of the two feature boundaries the code depends on, so it is the earliest Node version where everything the repo ships and tests runs unflagged: + +- **`node:sqlite` — Node 22.13.** `packages/session-persistence/session-persistence-sqlite` does a top-level `import { DatabaseSync } from 'node:sqlite'`. The module dropped its `--experimental-sqlite` flag requirement in Node 22.13 (backport of the 23.4 change), so any floor ≥ 22.13 loads it without a flag. +- **Native TypeScript type-stripping — Node 22.18.** The `packages/ui/stdio-agent/tests/built-bin.e2e.ts` smoke boots the published `lib/bin.js` under plain `node` and loads the example's `.ts` plugins (`mock-llm.ts`, `echo-tool.ts`) with no tsx. Native type-stripping — which makes that work — was unflagged in the 22.x LTS line only in 22.18 (before that it needed `--experimental-strip-types`). This is the binding constraint, so it sets the floor. + +`@types/node` is pinned to the 22.x line (`^22.20.0`) to match the floor: reaching for a Node 23+/24+/25+ API then fails `tsc` on every machine and in the typecheck gate, rather than compiling clean and surviving to a runtime failure only the 22.18 matrix leg could catch. The whole tree typechecks clean against the Node 22 type surface today, so the pin costs nothing. + +## Consequences + +- The supported base widens to the Node 22 LTS line, and the `['22.18', 24, 26]` matrix proves it on every push and PR rather than trusting the manifest. +- The built-bin smoke needs no version-conditional flag: at 22.18 type-stripping is already the default, so the test stays the plain `node lib/bin.js` path it documents. +- A future change reaching for a Node 23+ API fails `tsc` immediately (the `@types/node` pin); one reaching for an API added in 22.19/22.20 — inside the 22.x type surface but above the floor — is caught instead by the 22.18 matrix leg. Either way the floor must move in the same change. +- The `vendor/hmr` and `vendor/loader` comments about Node 24 module-cache internals are unaffected — they describe dev-time HMR loader behavior, not the shipped runtime contract, and are pinned vendored source. + +## Alternatives considered + +- **Floor `>=22.13` (the `node:sqlite` boundary) plus `--experimental-strip-types` in the built-bin smoke on 22.13–22.17.** Rejected: it adds a version-conditional test flag for one narrow range and dresses up an experimental-flag dependency as first-class support. 22.18 clears both boundaries with zero test special-casing, and the five-patch gap below it buys nothing real. +- **Keep `>=24`.** Rejected: it excludes Node 22 LTS with no runtime justification once the two boundaries above are known. +- **Matrix `[22, 24, 26]` (latest 22.x) instead of pinning `22.18`.** Rejected: "latest 22.x" drifts upward over time and would silently stop exercising the declared floor. Pinning the floor version is what makes the matrix a proof of the claim rather than a proof of some newer 22.x. +- **Keep `@types/node` ahead of the floor (`^25`).** Rejected: types ahead of the runtime floor let a Node 24/25-only API compile clean and fail only at runtime on 22.18 — exactly the "green types, broken product" gap. Pinning `@types/node` to the 22.x line turns that into a compile error everywhere, and the tree already typechecks clean against the Node 22 surface, so the pin is free. diff --git a/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md b/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md index 982c0fcb3c..5be4427b36 100644 --- a/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md +++ b/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md @@ -54,7 +54,7 @@ The repo secret is named `DEEPSEEK_API_KEY_EXTERNAL`; it is mapped to the `DEEPS ### Scope, runtime shape -Run **only** `test:e2e`. The keyless gates (typecheck/lint/coverage/snapshot/build/hygiene) already run in ci.yml on every push and PR; repeating them here would duplicate signal and slow the real-API job. No build step — e2e tests run unbuilt via tsx + the tsconfig paths map. Single Node 24 (the `engines` floor): these tests exercise *API integration*, not node-version compat, which ci.yml's Node 24/26 jobs already own; a second Node version would double real-API calls for no added signal. `vitest.e2e.config.ts` runs files through a bounded worker pool (`DSH_E2E_MAX_WORKERS`, default `4`, CI value `14`) so CI and local with-key runs parallelize independent files while retaining a one-line serial escape hatch for quota investigations. `timeout-minutes: 45` bounds a wedged run given 120s/test and `retry: 2`. `cancel-in-progress` is enabled only for `pull_request` runs — a superseded PR run is on a stale commit and worth cancelling, whereas a push/schedule run is already producing the post-merge/nightly signal and is never cancelled. +Run **only** `test:e2e`. The keyless gates (typecheck/lint/coverage/snapshot/build/hygiene) already run in ci.yml on every push and PR; repeating them here would duplicate signal and slow the real-API job. No build step — e2e tests run unbuilt via tsx + the tsconfig paths map. Node matrix `['22.18', 24]` (the `engines` floor plus the primary line): the keyless ci.yml matrix exercises only the MOCK adapter (`dsh-llm-replay`), so the real `fetch` + SSE-streaming adapter path — and its node-version compat — runs nowhere else. Running the real-API suite on both the floor and the primary line closes that gap; 26 is left to the keyless matrix, since floor + LTS is the meaningful pair for the live network path and inference is cheap (we are DeepSeek). `vitest.e2e.config.ts` runs files through a bounded worker pool (`DSH_E2E_MAX_WORKERS`, default `4`, CI value `14`) so CI and local with-key runs parallelize independent files while retaining a one-line serial escape hatch for quota investigations. `timeout-minutes: 45` bounds a wedged run given 120s/test and `retry: 2`. `cancel-in-progress` is enabled only for `pull_request` runs — a superseded PR run is on a stale commit and worth cancelling, whereas a push/schedule run is already producing the post-merge/nightly signal and is never cancelled. ## Security diff --git a/package.json b/package.json index dc8f487bd9..54ce68e408 100644 --- a/package.json +++ b/package.json @@ -5,7 +5,7 @@ "type": "module", "packageManager": "pnpm@11.7.0", "engines": { - "node": ">=24" + "node": ">=22.18" }, "workspaces": [ "vendor/*", @@ -70,7 +70,7 @@ "@stylistic/eslint-plugin": "^5.10.0", "@types/jsdom": "^28.0.3", "@types/mdast": "^4.0.4", - "@types/node": "^25.3.5", + "@types/node": "^22.20.0", "@vitest/coverage-v8": "^4.1.8", "eslint": "^10.4.1", "fast-check": "^4.8.0", diff --git a/packages/session-persistence/session-persistence-sqlite/README.md b/packages/session-persistence/session-persistence-sqlite/README.md index d7095633c9..89ab74f09c 100644 --- a/packages/session-persistence/session-persistence-sqlite/README.md +++ b/packages/session-persistence/session-persistence-sqlite/README.md @@ -8,7 +8,7 @@ A SQLite durable session-persistence backend — a second `SessionPersistence` i Each `SessionEvent` maps 1:1 onto a row in an `events` table `(session_id, seq, type, time, data, source_event_seqs, surface_op)` — `data` is the event payload as JSON text, so the row shape is the event verbatim (including `assistant/chunk`, keeping `seq` contiguous). The two `TEXT` columns `source_event_seqs` and `surface_op` are nullable; they store the event's optional surface-metadata fields (see [session surface](../../../docs/rfc/implemented/architecture/2026-06-18-session-surface.md)). Out-of-log metadata (`SessionHeader`) lives in a `sessions` row. A `sessions` row is written only by the first `append` — its existence is the lazy-materialization signal (`list` reports exactly the sessions that have a row), so no separate column is needed. -The repo targets Node ≥ 24 (the root `engines` field), which includes the stable `node:sqlite` module. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software). +The repo targets Node ≥ 22.18 (the root `engines` field), which includes `node:sqlite` unflagged — the module has been available without the `--experimental-sqlite` flag since Node 22.13, so this backend's top-level `import { DatabaseSync } from 'node:sqlite'` loads without a flag on every supported version. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software). ## Contract semantics over rows diff --git a/packages/ui/stdio-agent/tests/built-bin.e2e.ts b/packages/ui/stdio-agent/tests/built-bin.e2e.ts index cda8b41b1f..38e0170ff5 100644 --- a/packages/ui/stdio-agent/tests/built-bin.e2e.ts +++ b/packages/ui/stdio-agent/tests/built-bin.e2e.ts @@ -76,9 +76,10 @@ async function makeConsumer(welcome: string, disabledBrokenEntry = false): Promi await mkdir(dirname(target), { recursive: true }) await symlink(abs, target) } - // The example's mock model + echo tool are example-local TS plugins (Node 24+ - // strips types natively, so plain `node` loads them); they import the workspace - // packages the symlinked node_modules now provides. + // The example's mock model + echo tool are example-local TS plugins (Node + // 22.18+ — the engines floor — strips types natively, so plain `node` loads + // them); they import the workspace packages the symlinked node_modules now + // provides. await cp(join(repoRoot, 'examples/echo-agent/src'), join(dir, 'src'), { recursive: true }) await writeFile(join(dir, 'cordis.yml'), [ '- id: mock-llm', diff --git a/packages/web/web-fetch-local/src/provider.ts b/packages/web/web-fetch-local/src/provider.ts index 29b183b710..acfca02e78 100644 --- a/packages/web/web-fetch-local/src/provider.ts +++ b/packages/web/web-fetch-local/src/provider.ts @@ -1,6 +1,6 @@ /** * `LocalFetchProvider`: a `WebFetchProvider` that retrieves a concrete public - * HTTP(S) URL with the platform-native `fetch` (Node 24) and returns a status + * HTTP(S) URL with the platform-native `fetch` (Node 22.18) and returns a status * code plus bounded decoded content. It owns SAFE RESOURCE RETRIEVAL — URL * validation, redirect policy, timeout, abort, byte caps, charset decoding, * content-type classification, binary rejection — but NOT presentation diff --git a/packages/web/web-search-deepseek/src/provider.ts b/packages/web/web-search-deepseek/src/provider.ts index 40566b4f75..ade27721a4 100644 --- a/packages/web/web-search-deepseek/src/provider.ts +++ b/packages/web/web-search-deepseek/src/provider.ts @@ -12,7 +12,7 @@ * `web_search_tool_result` block (native search did not trigger), it throws * `WEB_PROVIDER_ERROR` rather than degrading to prose-scraping. * - * Network requests use platform-native `fetch` (Node 24), mirroring + * Network requests use platform-native `fetch` (Node 22.18), mirroring * `@deepseek-ai/dsh-llm-deepseek`'s adapter — not a cordis HTTP-client service. * The Anthropic wire shape is a provider-private detail and does NOT make this * provider depend on `ctx.llm`. diff --git a/packages/web/web-search-exa/src/provider.ts b/packages/web/web-search-exa/src/provider.ts index f187f90344..8e3f7d8b02 100644 --- a/packages/web/web-search-exa/src/provider.ts +++ b/packages/web/web-search-exa/src/provider.ts @@ -6,7 +6,7 @@ * `title`, the first highlight as `snippet`, and `publishedDate` as * `publishedAt`. * - * Network requests use platform-native `fetch` (Node 24), mirroring + * Network requests use platform-native `fetch` (Node 22.18), mirroring * `@deepseek-ai/dsh-llm-deepseek`'s adapter — not a cordis HTTP-client service. * * @module @deepseek-ai/dsh-web-search-exa/provider diff --git a/packages/web/web-search-perplexity/src/provider.ts b/packages/web/web-search-perplexity/src/provider.ts index ed72ea82c3..f4a12fb415 100644 --- a/packages/web/web-search-perplexity/src/provider.ts +++ b/packages/web/web-search-perplexity/src/provider.ts @@ -5,7 +5,7 @@ * structured `search_results[]` for `sources[]`, falling back to the URL-only * `citations[]` when `search_results` is absent. * - * Network requests use platform-native `fetch` (Node 24), mirroring + * Network requests use platform-native `fetch` (Node 22.18), mirroring * `@deepseek-ai/dsh-llm-deepseek`'s adapter. The OpenAI-compatible request shape * is a provider-private detail and does NOT make this provider depend on * `ctx.llm`. diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 97bcc8b288..0e68ee6fa9 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -21,8 +21,8 @@ importers: specifier: ^4.0.4 version: 4.0.4 '@types/node': - specifier: ^25.3.5 - version: 25.9.3 + specifier: ^22.20.0 + version: 22.20.0 '@vitest/coverage-v8': specifier: ^4.1.8 version: 4.1.8(vitest@4.1.8) @@ -70,10 +70,10 @@ importers: version: 8.61.0(eslint@10.5.0(jiti@2.7.0))(typescript@6.0.3) vite-tsconfig-paths: specifier: ^6.1.1 - version: 6.1.1(typescript@6.0.3)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + version: 6.1.1(typescript@6.0.3)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) vitest: specifier: ^4.1.8 - version: 4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + version: 4.1.8(@types/node@22.20.0)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) packages/bash/bash: devDependencies: @@ -2348,6 +2348,9 @@ packages: '@types/ms@2.1.0': resolution: {integrity: sha512-GsCCIZDE/p3i96vtEqx+7dBUGXrc7zeSK3wwPHIaRThS+9OhWIXRqzs4d6k1SVU8g91DrNRWxWUGhp5KXQb2VA==} + '@types/node@22.20.0': + resolution: {integrity: sha512-QWlFW2wf3nTjC13/DqRnBpR4ZO36VJH/JVBkA/vcnmbTBNQIlnObqyqZE1tUR7+Ni23Lda8R1BxMfbXRpCUx5g==} + '@types/node@25.9.3': resolution: {integrity: sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg==} @@ -3811,6 +3814,9 @@ packages: unconfig-core@7.5.0: resolution: {integrity: sha512-Su3FauozOGP44ZmKdHy2oE6LPjk51M/TRRjHv2HNCWiDvfvCoxC2lno6jevMA91MYAdCdwP05QnWdWpSbncX/w==} + undici-types@6.21.0: + resolution: {integrity: sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==} + undici-types@7.24.6: resolution: {integrity: sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==} @@ -5080,6 +5086,10 @@ snapshots: '@types/ms@2.1.0': {} + '@types/node@22.20.0': + dependencies: + undici-types: 6.21.0 + '@types/node@25.9.3': dependencies: undici-types: 7.24.6 @@ -5197,7 +5207,7 @@ snapshots: obug: 2.1.3 std-env: 4.1.0 tinyrainbow: 3.1.0 - vitest: 4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + vitest: 4.1.8(@types/node@22.20.0)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) '@vitest/expect@4.1.8': dependencies: @@ -5208,13 +5218,13 @@ snapshots: chai: 6.2.2 tinyrainbow: 3.1.0 - '@vitest/mocker@4.1.8(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0))': + '@vitest/mocker@4.1.8(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0))': dependencies: '@vitest/spy': 4.1.8 estree-walker: 3.0.3 magic-string: 0.30.21 optionalDependencies: - vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + vite: 8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) '@vitest/pretty-format@4.1.8': dependencies: @@ -6808,6 +6818,8 @@ snapshots: '@quansync/fs': 1.0.0 quansync: 1.0.0 + undici-types@6.21.0: {} + undici-types@7.24.6: {} undici@7.28.0: {} @@ -6837,17 +6849,17 @@ snapshots: uuid@14.0.1: {} - vite-tsconfig-paths@6.1.1(typescript@6.0.3)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): + vite-tsconfig-paths@6.1.1(typescript@6.0.3)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): dependencies: debug: 4.4.3 globrex: 0.1.2 tsconfck: 3.1.6(typescript@6.0.3) - vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + vite: 8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) transitivePeerDependencies: - supports-color - typescript - vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0): + vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0): dependencies: lightningcss: 1.32.0 picomatch: 4.0.4 @@ -6855,17 +6867,17 @@ snapshots: rolldown: 1.0.3 tinyglobby: 0.2.17 optionalDependencies: - '@types/node': 25.9.3 + '@types/node': 22.20.0 esbuild: 0.28.1 fsevents: 2.3.3 jiti: 2.7.0 tsx: 4.22.4 yaml: 2.9.0 - vitest@4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): + vitest@4.1.8(@types/node@22.20.0)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): dependencies: '@vitest/expect': 4.1.8 - '@vitest/mocker': 4.1.8(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + '@vitest/mocker': 4.1.8(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) '@vitest/pretty-format': 4.1.8 '@vitest/runner': 4.1.8 '@vitest/snapshot': 4.1.8 @@ -6882,10 +6894,10 @@ snapshots: tinyexec: 1.2.4 tinyglobby: 0.2.17 tinyrainbow: 3.1.0 - vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + vite: 8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) why-is-node-running: 2.3.0 optionalDependencies: - '@types/node': 25.9.3 + '@types/node': 22.20.0 '@vitest/coverage-v8': 4.1.8(vitest@4.1.8) jsdom: 29.1.1 transitivePeerDependencies: From 1c2823c73d751af5373b867cafd188c40bcf5ade Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Mon, 6 Jul 2026 12:16:31 +0800 Subject: [PATCH 07/24] fix(scripts): replace async fs glob with globSync (failed on Node 22.18) --- scripts/doc-typecheck.ts | 5 ++--- scripts/verify-doc-refs.ts | 5 ++--- scripts/verify-md-links.ts | 5 ++--- scripts/verify-md-wrap.ts | 5 ++--- scripts/verify-mermaid.ts | 5 ++--- scripts/verify-package-paths.ts | 5 ++--- scripts/verify-translation-pairing.ts | 5 ++--- scripts/verify-type-equiv.ts | 5 ++--- 8 files changed, 16 insertions(+), 24 deletions(-) diff --git a/scripts/doc-typecheck.ts b/scripts/doc-typecheck.ts index 6f40cccd0a..e57f3710ee 100644 --- a/scripts/doc-typecheck.ts +++ b/scripts/doc-typecheck.ts @@ -25,9 +25,8 @@ */ import { execFileSync } from 'node:child_process' -import { mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { globSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs' import { join, relative, resolve } from 'node:path' -import { glob } from 'node:fs/promises' import ts from 'typescript' const root = resolve(import.meta.dirname, '..') @@ -139,7 +138,7 @@ const markdownGlobs = ['README.md', 'docs/**/*.md', 'packages/*/*.md', 'packages const files: string[] = [] for (const pattern of markdownGlobs) { - for await (const match of glob(pattern, { cwd: root })) files.push(resolve(root, match)) + for (const match of globSync(pattern, { cwd: root })) files.push(resolve(root, match)) } files.sort() diff --git a/scripts/verify-doc-refs.ts b/scripts/verify-doc-refs.ts index a53399a272..56be5140c8 100644 --- a/scripts/verify-doc-refs.ts +++ b/scripts/verify-doc-refs.ts @@ -26,9 +26,8 @@ * Run: `tsx scripts/verify-doc-refs.ts`. */ -import { existsSync, readFileSync } from 'node:fs' +import { existsSync, globSync, readFileSync } from 'node:fs' import { relative, resolve } from 'node:path' -import { glob } from 'node:fs/promises' const root = resolve(import.meta.dirname, '..') @@ -77,7 +76,7 @@ function findViolations(absPath: string): Violation[] { const all: Violation[] = [] let checked = 0 for (const pattern of PATTERNS) { - for await (const match of glob(pattern, { cwd: root })) { + for (const match of globSync(pattern, { cwd: root })) { if (isExcluded(match)) continue checked++ all.push(...findViolations(resolve(root, match))) diff --git a/scripts/verify-md-links.ts b/scripts/verify-md-links.ts index 2a96cfd0af..d3e80e285e 100644 --- a/scripts/verify-md-links.ts +++ b/scripts/verify-md-links.ts @@ -32,9 +32,8 @@ * Run: `tsx scripts/verify-md-links.ts`. */ -import { existsSync, readFileSync, realpathSync } from 'node:fs' +import { existsSync, globSync, readFileSync, realpathSync } from 'node:fs' import { dirname, relative, resolve } from 'node:path' -import { glob } from 'node:fs/promises' import { fromMarkdown } from 'mdast-util-from-markdown' import { gfmFromMarkdown } from 'mdast-util-gfm' import { gfm } from 'micromark-extension-gfm' @@ -134,7 +133,7 @@ const seen = new Set() const all: Violation[] = [] let checked = 0 for (const pattern of PATTERNS) { - for await (const match of glob(pattern, { cwd: root })) { + for (const match of globSync(pattern, { cwd: root })) { const abs = resolve(root, match) // CLAUDE.md symlinks resolve onto AGENTS.md; dedupe by real path so a file // matched twice (or via symlink) is checked once. diff --git a/scripts/verify-md-wrap.ts b/scripts/verify-md-wrap.ts index 3ffad3be43..2d8845e030 100644 --- a/scripts/verify-md-wrap.ts +++ b/scripts/verify-md-wrap.ts @@ -25,9 +25,8 @@ * Run: `tsx scripts/verify-md-wrap.ts`. */ -import { readFileSync, realpathSync } from 'node:fs' +import { globSync, readFileSync, realpathSync } from 'node:fs' import { relative, resolve } from 'node:path' -import { glob } from 'node:fs/promises' import { fromMarkdown } from 'mdast-util-from-markdown' import { gfmFromMarkdown } from 'mdast-util-gfm' import { gfm } from 'micromark-extension-gfm' @@ -76,7 +75,7 @@ const seen = new Set() const all: Violation[] = [] let checked = 0 for (const pattern of PATTERNS) { - for await (const match of glob(pattern, { cwd: root })) { + for (const match of globSync(pattern, { cwd: root })) { const abs = resolve(root, match) // CLAUDE.md symlinks resolve onto AGENTS.md; dedupe by real path so a file // matched twice (or via symlink) is checked once. diff --git a/scripts/verify-mermaid.ts b/scripts/verify-mermaid.ts index 954c246640..3f9af495b3 100644 --- a/scripts/verify-mermaid.ts +++ b/scripts/verify-mermaid.ts @@ -12,9 +12,8 @@ * Run: `tsx scripts/verify-mermaid.ts`. */ -import { readFileSync, realpathSync } from 'node:fs' +import { globSync, readFileSync, realpathSync } from 'node:fs' import { resolve } from 'node:path' -import { glob } from 'node:fs/promises' import { fromMarkdown } from 'mdast-util-from-markdown' import { gfmFromMarkdown } from 'mdast-util-gfm' import { gfm } from 'micromark-extension-gfm' @@ -72,7 +71,7 @@ const blocks: Block[] = [] const seen = new Set() let checkedFiles = 0 for (const pattern of PATTERNS) { - for await (const match of glob(pattern, { cwd: root })) { + for (const match of globSync(pattern, { cwd: root })) { const real = realpathSync(resolve(root, match)) if (seen.has(real)) continue seen.add(real) diff --git a/scripts/verify-package-paths.ts b/scripts/verify-package-paths.ts index 7bec754dba..5d2d91982b 100644 --- a/scripts/verify-package-paths.ts +++ b/scripts/verify-package-paths.ts @@ -39,9 +39,8 @@ * Run: `tsx scripts/verify-package-paths.ts`. */ -import { existsSync, readdirSync, readFileSync, realpathSync } from 'node:fs' +import { existsSync, globSync, readdirSync, readFileSync, realpathSync } from 'node:fs' import { relative, resolve } from 'node:path' -import { glob } from 'node:fs/promises' const root = resolve(import.meta.dirname, '..') @@ -144,7 +143,7 @@ const all: Violation[] = [] let checked = 0 const seen = new Set() for (const pattern of PATTERNS) { - for await (const match of glob(pattern, { cwd: root })) { + for (const match of globSync(pattern, { cwd: root })) { if (isExcluded(match)) continue // Dedup by real path: the root/packages CLAUDE.md are symlinks to AGENTS.md. const real = realpathSync(resolve(root, match)) diff --git a/scripts/verify-translation-pairing.ts b/scripts/verify-translation-pairing.ts index eea30e4b22..3d80572e7c 100644 --- a/scripts/verify-translation-pairing.ts +++ b/scripts/verify-translation-pairing.ts @@ -44,9 +44,8 @@ */ import { createHash } from 'node:crypto' -import { existsSync, readFileSync, writeFileSync } from 'node:fs' +import { existsSync, globSync, readFileSync, writeFileSync } from 'node:fs' import { basename, join, resolve } from 'node:path' -import { glob } from 'node:fs/promises' import { fromMarkdown } from 'mdast-util-from-markdown' import { gfmFromMarkdown } from 'mdast-util-gfm' import { gfm } from 'micromark-extension-gfm' @@ -211,7 +210,7 @@ function parse(content: string): Nodes { // Enumerate the scope once. const files = new Set() for (const pattern of SCOPE_PATTERNS) { - for await (const match of glob(pattern, { cwd: root })) files.add(match) + for (const match of globSync(pattern, { cwd: root })) files.add(match) } const translations = [...files].filter(f => f.endsWith('.zh.md')).sort() const metas = [...files].filter(f => f.endsWith('.i18n.yaml')).sort() diff --git a/scripts/verify-type-equiv.ts b/scripts/verify-type-equiv.ts index c93383e40f..85ccd642d9 100644 --- a/scripts/verify-type-equiv.ts +++ b/scripts/verify-type-equiv.ts @@ -22,9 +22,8 @@ * Run: `tsx scripts/verify-type-equiv.ts`. */ -import { readFileSync, existsSync } from 'node:fs' +import { globSync, readFileSync, existsSync } from 'node:fs' import { resolve } from 'node:path' -import { glob } from 'node:fs/promises' import ts from 'typescript' const root = resolve(import.meta.dirname, '..') @@ -148,7 +147,7 @@ const keyOf = (x: { doc: string; symbol: string }): string => `${x.doc}::${x.sym // as an orphan rather than silently skipped. const docSet = new Set() for (const pattern of MARKDOWN_GLOBS) { - for await (const match of glob(pattern, { cwd: root })) docSet.add(match) + for (const match of globSync(pattern, { cwd: root })) docSet.add(match) } const blocks: EquivBlock[] = [...docSet].sort().flatMap(extractEquivBlocks) From 393da2b9836a28eb3ceb43e4ea67a8e9cecc5451 Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Mon, 6 Jul 2026 13:45:13 +0800 Subject: [PATCH 08/24] =?UTF-8?q?fix:=20engines=20^22.18.0=20||=20>=3D24.0?= =?UTF-8?q?.0=20=E2=80=94=20exclude=20EOL=20Node=2023?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- AGENTS.md | 2 +- docs/development.i18n.yaml | 4 ++-- docs/development.md | 2 +- docs/development.zh.md | 2 +- .../implemented/process/2026-07-06-node-22-18-floor.md | 10 +++++++--- package.json | 2 +- .../session-persistence-sqlite/README.md | 2 +- 7 files changed, 14 insertions(+), 10 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index c97ca15b55..6f83dae519 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -34,7 +34,7 @@ Per-package map: the group READMEs, indexed from [packages/README.md](packages/R ## Commands ```sh -pnpm install # pnpm workspaces, node >= 22.18 +pnpm install # pnpm workspaces, node ^22.18 || >=24 pnpm run test # vitest unit tests pnpm run test:coverage # THE gating test run: per-file 100% coverage on packages/*/*/src pnpm run test:e2e # real-API tests; self-skip without DEEPSEEK_API_KEY diff --git a/docs/development.i18n.yaml b/docs/development.i18n.yaml index 2fc2c3cfe9..9b0bbbd1d6 100644 --- a/docs/development.i18n.yaml +++ b/docs/development.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -development.md: 8f901909264d4405d396782d68c28fbde9b85bfa -development.zh.md: 2b2af080d11b8ea9d9cbf26c62833b4fc00fb6ba +development.md: 3acbff05204e6c7f44c6a8a727052a6abf881fab +development.zh.md: 5a02cce00c26baf5c94d4a20a9138c5af89c27c3 diff --git a/docs/development.md b/docs/development.md index 8f90190926..3acbff0520 100644 --- a/docs/development.md +++ b/docs/development.md @@ -6,7 +6,7 @@ This guide covers the local setup needed to work on DeepSeek Harness and underst ## Prerequisites -- Node.js 22.18 or newer. The repo declares `node >=22.18`; CI runs the matrix on Node 22.18, 24, and 26. +- Node.js `^22.18.0 || >=24.0.0` (22.18+ on the LTS line, or 24+). The Node 23 line is excluded: `node:sqlite` (until 23.4) and native TS type-stripping (until 23.6) are still flagged there, and 23 is non-LTS/EOL. CI runs the matrix on Node 22.18, 24, and 26. - Corepack-enabled pnpm. The repo pins `pnpm@11.7.0` in `package.json`; run `corepack enable` if `pnpm --version` does not resolve through Corepack. - Git. - Optional: a DeepSeek API key for the REPL/ACP agent demos and real-API e2e tests. diff --git a/docs/development.zh.md b/docs/development.zh.md index 2b2af080d1..5a02cce00c 100644 --- a/docs/development.zh.md +++ b/docs/development.zh.md @@ -6,7 +6,7 @@ ## 前置条件 -- Node.js 22.18 或更新版本。仓库声明 `node >=22.18`;CI 在 Node 22.18、24 和 26 上跑矩阵。 +- Node.js `^22.18.0 || >=24.0.0`(即 LTS 线的 22.18+,或 24+)。排除 Node 23 线:那里 `node:sqlite`(要到 23.4)和原生 TS 类型剥离(要到 23.6)仍需 flag,且 23 是非 LTS、已 EOL。CI 在 Node 22.18、24 和 26 上跑矩阵。 - 启用了 Corepack 的 pnpm。仓库在 `package.json` 中钉住 `pnpm@11.7.0`;如果 `pnpm --version` 无法通过 Corepack 解析,先运行 `corepack enable`。 - Git。 - 可选:一个 DeepSeek API key,用于 REPL/ACP agent(智能体)演示和真实 API 的 e2e 测试。 diff --git a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md index 737bb2e40f..128d06246f 100644 --- a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md +++ b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md @@ -8,10 +8,12 @@ The root `engines.node` was `>=24`, which excluded the entire Node 22 LTS line f ## Decision -Set `engines.node` to `>=22.18` and treat 22.18 as the tested floor everywhere (CI matrix `['22.18', 24, 26]`, the real-API e2e job on `['22.18', 24]` — floor plus primary line, since the keyless matrix exercises only the mock adapter and the live `fetch`/SSE path runs only in e2e). 22.18 is the *later* of the two feature boundaries the code depends on, so it is the earliest Node version where everything the repo ships and tests runs unflagged: +Set `engines.node` to `^22.18.0 || >=24.0.0` (Node 22.18+ on the LTS line, or 24+) and test it on the CI matrix `['22.18', 24, 26]` — the real-API e2e job on `['22.18', 24]` (floor plus primary line, since the keyless matrix exercises only the mock adapter and the live `fetch`/SSE path runs only in e2e). Two Node features gate the range, each with its own LTS-line and Current-line unflag point: -- **`node:sqlite` — Node 22.13.** `packages/session-persistence/session-persistence-sqlite` does a top-level `import { DatabaseSync } from 'node:sqlite'`. The module dropped its `--experimental-sqlite` flag requirement in Node 22.13 (backport of the 23.4 change), so any floor ≥ 22.13 loads it without a flag. -- **Native TypeScript type-stripping — Node 22.18.** The `packages/ui/stdio-agent/tests/built-bin.e2e.ts` smoke boots the published `lib/bin.js` under plain `node` and loads the example's `.ts` plugins (`mock-llm.ts`, `echo-tool.ts`) with no tsx. Native type-stripping — which makes that work — was unflagged in the 22.x LTS line only in 22.18 (before that it needed `--experimental-strip-types`). This is the binding constraint, so it sets the floor. +- **`node:sqlite`** — `packages/session-persistence/session-persistence-sqlite` does a top-level `import { DatabaseSync } from 'node:sqlite'`. The module dropped its `--experimental-sqlite` flag requirement at **22.13** (LTS) and **23.4** (Current); before those, importing it throws at load. +- **Native TypeScript type-stripping** — the `packages/ui/stdio-agent/tests/built-bin.e2e.ts` smoke boots the published `lib/bin.js` under plain `node` (no tsx) and loads the example's `.ts` plugins (`mock-llm.ts`, `echo-tool.ts`). Type-stripping is the default from **22.18** (LTS) and **23.6** (Current); before those it needs `--experimental-strip-types`. + +On the 22.x line both features clear at **22.18** (the later of 22.13/22.18), so `^22.18.0` is the LTS floor. The range is **disjoint** rather than an open `>=22.18` because the Node **23.0–23.5** window still has at least one feature flagged (sqlite until 23.4, stripping until 23.6): `>=22.18` would advertise support there, where the sqlite backend throws `ERR_UNKNOWN_BUILTIN_MODULE` at load. Node 23 is non-LTS and already end-of-life, so rather than carve out `>=23.6` the range skips the whole line and resumes at `>=24.0.0` — the same shape several of the repo's own dependencies already declare (`^22.18.0 || >=24.11.0`). `@types/node` is pinned to the 22.x line (`^22.20.0`) to match the floor: reaching for a Node 23+/24+/25+ API then fails `tsc` on every machine and in the typecheck gate, rather than compiling clean and surviving to a runtime failure only the 22.18 matrix leg could catch. The whole tree typechecks clean against the Node 22 type surface today, so the pin costs nothing. @@ -26,5 +28,7 @@ Set `engines.node` to `>=22.18` and treat 22.18 as the tested floor everywhere ( - **Floor `>=22.13` (the `node:sqlite` boundary) plus `--experimental-strip-types` in the built-bin smoke on 22.13–22.17.** Rejected: it adds a version-conditional test flag for one narrow range and dresses up an experimental-flag dependency as first-class support. 22.18 clears both boundaries with zero test special-casing, and the five-patch gap below it buys nothing real. - **Keep `>=24`.** Rejected: it excludes Node 22 LTS with no runtime justification once the two boundaries above are known. +- **Open-ended `>=22.18`.** Rejected: it advertises support for Node 23.0–23.5, where `node:sqlite` (until 23.4) or type-stripping (until 23.6) is still flagged, so the sqlite backend throws at load. The disjoint `^22.18.0 || >=24.0.0` matches the real runtime boundary. +- **Include Node 23.6+ (`^22.18.0 || >=23.6.0`).** Rejected: 23.6+ does run both features unflagged, but Node 23 is end-of-life — advertising a dead release line adds a range term (and, to back it, a CI leg) for a runtime no deployment should use. 24 is the meaningful resumption point, and the 22.18 and 24 legs already bracket the same unflagged code paths. - **Matrix `[22, 24, 26]` (latest 22.x) instead of pinning `22.18`.** Rejected: "latest 22.x" drifts upward over time and would silently stop exercising the declared floor. Pinning the floor version is what makes the matrix a proof of the claim rather than a proof of some newer 22.x. - **Keep `@types/node` ahead of the floor (`^25`).** Rejected: types ahead of the runtime floor let a Node 24/25-only API compile clean and fail only at runtime on 22.18 — exactly the "green types, broken product" gap. Pinning `@types/node` to the 22.x line turns that into a compile error everywhere, and the tree already typechecks clean against the Node 22 surface, so the pin is free. diff --git a/package.json b/package.json index 54ce68e408..60770f0d1f 100644 --- a/package.json +++ b/package.json @@ -5,7 +5,7 @@ "type": "module", "packageManager": "pnpm@11.7.0", "engines": { - "node": ">=22.18" + "node": "^22.18.0 || >=24.0.0" }, "workspaces": [ "vendor/*", diff --git a/packages/session-persistence/session-persistence-sqlite/README.md b/packages/session-persistence/session-persistence-sqlite/README.md index 89ab74f09c..f3601140d2 100644 --- a/packages/session-persistence/session-persistence-sqlite/README.md +++ b/packages/session-persistence/session-persistence-sqlite/README.md @@ -8,7 +8,7 @@ A SQLite durable session-persistence backend — a second `SessionPersistence` i Each `SessionEvent` maps 1:1 onto a row in an `events` table `(session_id, seq, type, time, data, source_event_seqs, surface_op)` — `data` is the event payload as JSON text, so the row shape is the event verbatim (including `assistant/chunk`, keeping `seq` contiguous). The two `TEXT` columns `source_event_seqs` and `surface_op` are nullable; they store the event's optional surface-metadata fields (see [session surface](../../../docs/rfc/implemented/architecture/2026-06-18-session-surface.md)). Out-of-log metadata (`SessionHeader`) lives in a `sessions` row. A `sessions` row is written only by the first `append` — its existence is the lazy-materialization signal (`list` reports exactly the sessions that have a row), so no separate column is needed. -The repo targets Node ≥ 22.18 (the root `engines` field), which includes `node:sqlite` unflagged — the module has been available without the `--experimental-sqlite` flag since Node 22.13, so this backend's top-level `import { DatabaseSync } from 'node:sqlite'` loads without a flag on every supported version. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software). +The repo's `engines.node` is `^22.18.0 || >=24.0.0` (Node 22.18+ or 24+). `node:sqlite` ships without the `--experimental-sqlite` flag from Node 22.13 (LTS) and 23.4 / 24 (Current) on; the range deliberately excludes the Node 23.0–23.3 window, where the module is still flagged and this backend's top-level `import { DatabaseSync } from 'node:sqlite'` would throw at load. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software). ## Contract semantics over rows From 6edce91735423968efe7633b6468ecc86efb41fa Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Tue, 7 Jul 2026 17:24:57 +0800 Subject: [PATCH 09/24] ci: e2e stay on Node 24 --- .github/workflows/e2e.yml | 14 ++------------ .../process/2026-07-06-node-22-18-floor.md | 2 +- .../testing/2026-06-19-real-api-e2e-ci.md | 2 +- 3 files changed, 4 insertions(+), 14 deletions(-) diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index eb73308b50..c0371947fd 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -49,17 +49,7 @@ permissions: jobs: e2e: runs-on: ubuntu-latest - strategy: - fail-fast: false - matrix: - # The keyless ci.yml matrix runs the MOCK adapter (dsh-llm-replay); the - # real fetch + SSE-streaming adapter path runs ONLY here, so its - # node-version compat is covered nowhere else. Run the real-API suite on - # the engines floor AND the primary line to close that gap. 26 is left to - # the keyless matrix — floor + LTS is the meaningful pair for the live - # network path, and inference is cheap (we are DeepSeek). - node: ['22.18', 24] - name: e2e node ${{ matrix.node }} + name: e2e # Run on every trusted event. Skip untrusted PRs (forks + Dependabot) where # the secret is withheld — they would otherwise hard-fail the preflight. if: >- @@ -73,7 +63,7 @@ jobs: - uses: actions/setup-node@v6 with: - node-version: ${{ matrix.node }} + node-version: 24 - name: Enable corepack (pnpm) run: corepack enable diff --git a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md index 128d06246f..8e0c736fd9 100644 --- a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md +++ b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md @@ -8,7 +8,7 @@ The root `engines.node` was `>=24`, which excluded the entire Node 22 LTS line f ## Decision -Set `engines.node` to `^22.18.0 || >=24.0.0` (Node 22.18+ on the LTS line, or 24+) and test it on the CI matrix `['22.18', 24, 26]` — the real-API e2e job on `['22.18', 24]` (floor plus primary line, since the keyless matrix exercises only the mock adapter and the live `fetch`/SSE path runs only in e2e). Two Node features gate the range, each with its own LTS-line and Current-line unflag point: +Set `engines.node` to `^22.18.0 || >=24.0.0` (Node 22.18+ on the LTS line, or 24+) and test it on the keyless CI matrix `['22.18', 24, 26]`. The real-API e2e workflow stays on Node 24 because it exercises API integration rather than the runtime floor. Two Node features gate the range, each with its own LTS-line and Current-line unflag point: - **`node:sqlite`** — `packages/session-persistence/session-persistence-sqlite` does a top-level `import { DatabaseSync } from 'node:sqlite'`. The module dropped its `--experimental-sqlite` flag requirement at **22.13** (LTS) and **23.4** (Current); before those, importing it throws at load. - **Native TypeScript type-stripping** — the `packages/ui/stdio-agent/tests/built-bin.e2e.ts` smoke boots the published `lib/bin.js` under plain `node` (no tsx) and loads the example's `.ts` plugins (`mock-llm.ts`, `echo-tool.ts`). Type-stripping is the default from **22.18** (LTS) and **23.6** (Current); before those it needs `--experimental-strip-types`. diff --git a/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md b/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md index 5be4427b36..1b348b1515 100644 --- a/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md +++ b/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md @@ -54,7 +54,7 @@ The repo secret is named `DEEPSEEK_API_KEY_EXTERNAL`; it is mapped to the `DEEPS ### Scope, runtime shape -Run **only** `test:e2e`. The keyless gates (typecheck/lint/coverage/snapshot/build/hygiene) already run in ci.yml on every push and PR; repeating them here would duplicate signal and slow the real-API job. No build step — e2e tests run unbuilt via tsx + the tsconfig paths map. Node matrix `['22.18', 24]` (the `engines` floor plus the primary line): the keyless ci.yml matrix exercises only the MOCK adapter (`dsh-llm-replay`), so the real `fetch` + SSE-streaming adapter path — and its node-version compat — runs nowhere else. Running the real-API suite on both the floor and the primary line closes that gap; 26 is left to the keyless matrix, since floor + LTS is the meaningful pair for the live network path and inference is cheap (we are DeepSeek). `vitest.e2e.config.ts` runs files through a bounded worker pool (`DSH_E2E_MAX_WORKERS`, default `4`, CI value `14`) so CI and local with-key runs parallelize independent files while retaining a one-line serial escape hatch for quota investigations. `timeout-minutes: 45` bounds a wedged run given 120s/test and `retry: 2`. `cancel-in-progress` is enabled only for `pull_request` runs — a superseded PR run is on a stale commit and worth cancelling, whereas a push/schedule run is already producing the post-merge/nightly signal and is never cancelled. +Run **only** `test:e2e`. The keyless gates (typecheck/lint/coverage/snapshot/build/hygiene) already run in ci.yml on every push and PR; repeating them here would duplicate signal and slow the real-API job. No build step — e2e tests run unbuilt via tsx + the tsconfig paths map. Single Node 24 (the primary line): these tests exercise API integration, not node-version compatibility, which ci.yml's Node 22.18/24/26 matrix owns. `vitest.e2e.config.ts` runs files through a bounded worker pool (`DSH_E2E_MAX_WORKERS`, default `4`, CI value `14`) so CI and local with-key runs parallelize independent files while retaining a one-line serial escape hatch for quota investigations. `timeout-minutes: 45` bounds a wedged run given 120s/test and `retry: 2`. `cancel-in-progress` is enabled only for `pull_request` runs — a superseded PR run is on a stale commit and worth cancelling, whereas a push/schedule run is already producing the post-merge/nightly signal and is never cancelled. ## Security From 92b5eccc961e350ca6a543453d6ac9661708f5eb Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Tue, 7 Jul 2026 17:39:04 +0800 Subject: [PATCH 10/24] build: upgrade to 22.19 for deps --- .github/workflows/ci.yml | 2 +- AGENTS.md | 2 +- docs/core-data-structures/web.md | 2 +- docs/development.i18n.yaml | 4 +- docs/development.md | 6 +-- docs/development.zh.md | 6 +-- docs/rfc/INDEX.md | 2 +- .../2026-06-24-web-capability-seam.md | 2 +- .../process/2026-06-11-quality-gates.md | 2 +- .../process/2026-07-06-node-22-18-floor.md | 34 ----------------- .../process/2026-07-06-node-engine-floor.md | 37 +++++++++++++++++++ .../testing/2026-06-19-real-api-e2e-ci.md | 2 +- package.json | 2 +- .../session-persistence-sqlite/README.md | 2 +- .../ui/stdio-agent/tests/built-bin.e2e.ts | 2 +- packages/web/web-fetch-local/src/provider.ts | 2 +- .../web/web-search-deepseek/src/provider.ts | 2 +- packages/web/web-search-exa/src/provider.ts | 2 +- .../web/web-search-perplexity/src/provider.ts | 2 +- 19 files changed, 59 insertions(+), 56 deletions(-) delete mode 100644 docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md create mode 100644 docs/rfc/implemented/process/2026-07-06-node-engine-floor.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 52d1ce75fa..e15f344653 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -101,7 +101,7 @@ jobs: strategy: fail-fast: false matrix: - node: ['22.18', 24, 26] + node: ['22.19', 24, 26] steps: - uses: actions/checkout@v6 diff --git a/AGENTS.md b/AGENTS.md index 6f83dae519..a6331c0b89 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -34,7 +34,7 @@ Per-package map: the group READMEs, indexed from [packages/README.md](packages/R ## Commands ```sh -pnpm install # pnpm workspaces, node ^22.18 || >=24 +pnpm install # pnpm workspaces, node ^22.19 || >=24 pnpm run test # vitest unit tests pnpm run test:coverage # THE gating test run: per-file 100% coverage on packages/*/*/src pnpm run test:e2e # real-API tests; self-skip without DEEPSEEK_API_KEY diff --git a/docs/core-data-structures/web.md b/docs/core-data-structures/web.md index 00f10278f2..f6bd6406ab 100644 --- a/docs/core-data-structures/web.md +++ b/docs/core-data-structures/web.md @@ -91,4 +91,4 @@ Selection never depends on registration, config, or HMR order: a capability has ## The service -`WebService` (`ctx.web`, defined in [`packages/web/web/src/index.ts`](../../packages/web/web/src/index.ts)) is a provider registry plus a provider-selecting execution surface, close to `LlmService`'s shape: `registerSearchProvider`/`registerFetchProvider` (duplicate ids throw `WEB_DUPLICATE_PROVIDER`, return disposers) and `search`/`fetch` (resolve the provider at call time, throw a structured `WebError` when the capability cannot run). Providers issue requests with the platform-native `fetch` (Node 22.18), mirroring `dsh-llm-deepseek`; the `dsh-web-fetch-local` provider owns safe retrieval (http/https-only, credential rejection, byte/char/timeout/redirect caps, same-origin-only redirects with per-hop re-validation, charset decoding) while `dsh-tool-web` owns presentation (HTML→markdown). SSRF / private-network blocking is deferred (see the RFC) — until it lands, `web_fetch` must not be enabled where it can reach sensitive internal targets. +`WebService` (`ctx.web`, defined in [`packages/web/web/src/index.ts`](../../packages/web/web/src/index.ts)) is a provider registry plus a provider-selecting execution surface, close to `LlmService`'s shape: `registerSearchProvider`/`registerFetchProvider` (duplicate ids throw `WEB_DUPLICATE_PROVIDER`, return disposers) and `search`/`fetch` (resolve the provider at call time, throw a structured `WebError` when the capability cannot run). Providers issue requests with platform-native `fetch` at the repo's Node floor, mirroring `dsh-llm-deepseek`; the `dsh-web-fetch-local` provider owns safe retrieval (http/https-only, credential rejection, byte/char/timeout/redirect caps, same-origin-only redirects with per-hop re-validation, charset decoding) while `dsh-tool-web` owns presentation (HTML→markdown). SSRF / private-network blocking is deferred (see the RFC) — until it lands, `web_fetch` must not be enabled where it can reach sensitive internal targets. diff --git a/docs/development.i18n.yaml b/docs/development.i18n.yaml index 9b0bbbd1d6..c3926e69bd 100644 --- a/docs/development.i18n.yaml +++ b/docs/development.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -development.md: 3acbff05204e6c7f44c6a8a727052a6abf881fab -development.zh.md: 5a02cce00c26baf5c94d4a20a9138c5af89c27c3 +development.md: 3cb96968ccbe291c3cd94937c88cd414f18f2166 +development.zh.md: 98f3ad4cd12ecd73fd8e23ad48278d8bc23d2496 diff --git a/docs/development.md b/docs/development.md index 3acbff0520..3cb96968cc 100644 --- a/docs/development.md +++ b/docs/development.md @@ -6,7 +6,7 @@ This guide covers the local setup needed to work on DeepSeek Harness and underst ## Prerequisites -- Node.js `^22.18.0 || >=24.0.0` (22.18+ on the LTS line, or 24+). The Node 23 line is excluded: `node:sqlite` (until 23.4) and native TS type-stripping (until 23.6) are still flagged there, and 23 is non-LTS/EOL. CI runs the matrix on Node 22.18, 24, and 26. +- Node.js `^22.19.0 || >=24.0.0` (22.19+ on the LTS line, or 24+). The LTS floor matches `@earendil-works/pi-ai`'s Node 22.19 dependency floor. The Node 23 line is excluded: `node:sqlite` (until 23.4) and native TS type-stripping (until 23.6) are still flagged there, and 23 is non-LTS/EOL. CI runs the compatibility matrix on Node 22.19, 24, and 26. - Corepack-enabled pnpm. The repo pins `pnpm@11.7.0` in `package.json`; run `corepack enable` if `pnpm --version` does not resolve through Corepack. - Git. - Optional: a DeepSeek API key for the REPL/ACP agent demos and real-API e2e tests. @@ -63,11 +63,11 @@ lefthook is configured in `lefthook.yml` as an early local checkpoint before rev The vendor manifest guard checks that changes under `vendor/*/src` are staged with the matching `vendor/README.md` manifest update. See `vendor/README.md` before editing vendored code. -These hooks do not exactly mirror CI. Notably, `pre-push` runs unit tests without coverage, while CI runs `pnpm run test:coverage`; CI also runs echo-agent and built-bin smoke tests and exercises the matrix on Node 22.18, 24, and 26. +These hooks do not exactly mirror CI. Notably, `pre-push` runs unit tests without coverage, while CI runs `pnpm run test:coverage`; CI also runs echo-agent and built-bin smoke tests and exercises the compatibility matrix on Node 22.19, 24, and 26. ## CI gates -The keyless GitHub workflow has six jobs: five Node 24 lanes run static gates, lint, coverage, snapshot replay, and artifact gates separately, and the Node 26 compatibility job runs `pnpm run check:node-compat`. The lane schedulers fan out independent gates from `package.json`: constraints, typecheck, lint, coverage, snapshot replay, `doc-sync` members, module-graph freshness, `knip`, and the echo-agent smoke test. +The keyless GitHub workflow has eight jobs: five Node 24 lanes run static gates, lint, coverage, snapshot replay, and artifact gates separately, and three compatibility jobs run `pnpm run check:node-compat` on Node 22.19, 24, and 26. The lane schedulers fan out independent gates from `package.json`: constraints, typecheck, lint, coverage, snapshot replay, `doc-sync` members, module-graph freshness, `knip`, and the echo-agent smoke test. `pnpm run build` feeds the artifact lane, and `publint`, `verify-node-next-types`, and built-bin smoke tests wait for build output. The separate real-API workflow runs `pnpm run test:e2e` with a secret and `DSH_E2E_MAX_WORKERS=14`. diff --git a/docs/development.zh.md b/docs/development.zh.md index 5a02cce00c..98f3ad4cd1 100644 --- a/docs/development.zh.md +++ b/docs/development.zh.md @@ -6,7 +6,7 @@ ## 前置条件 -- Node.js `^22.18.0 || >=24.0.0`(即 LTS 线的 22.18+,或 24+)。排除 Node 23 线:那里 `node:sqlite`(要到 23.4)和原生 TS 类型剥离(要到 23.6)仍需 flag,且 23 是非 LTS、已 EOL。CI 在 Node 22.18、24 和 26 上跑矩阵。 +- Node.js `^22.19.0 || >=24.0.0`(即 LTS 线的 22.19+,或 24+)。LTS floor 匹配 `@earendil-works/pi-ai` 的 Node 22.19 依赖 floor。排除 Node 23 线:那里 `node:sqlite`(要到 23.4)和原生 TS 类型剥离(要到 23.6)仍需 flag,且 23 是非 LTS、已 EOL。CI 在 Node 22.19、24 和 26 上跑兼容性矩阵。 - 启用了 Corepack 的 pnpm。仓库在 `package.json` 中钉住 `pnpm@11.7.0`;如果 `pnpm --version` 无法通过 Corepack 解析,先运行 `corepack enable`。 - Git。 - 可选:一个 DeepSeek API key,用于 REPL/ACP agent(智能体)演示和真实 API 的 e2e 测试。 @@ -63,11 +63,11 @@ lefthook 在 `lefthook.yml` 中配置,作为评审前的本地早期检查点 vendor manifest 守卫检查 `vendor/*/src` 下的改动是否连同对应的 `vendor/README.md` manifest 更新一起暂存。编辑 vendor 代码前先看 `vendor/README.md`。 -这些钩子并不与 CI 完全一致。特别是:`pre-push` 跑不带覆盖率的单元测试,而 CI 跑 `pnpm run test:coverage`;CI 还会跑 echo-agent 和 built-bin 冒烟测试,并在 Node 22.18、24 和 26 上跑矩阵。 +这些钩子并不与 CI 完全一致。特别是:`pre-push` 跑不带覆盖率的单元测试,而 CI 跑 `pnpm run test:coverage`;CI 还会跑 echo-agent 和 built-bin 冒烟测试,并在 Node 22.19、24 和 26 上跑兼容性矩阵。 ## CI 门禁 -keyless GitHub 工作流有六个 job:五个 Node 24 lane 分别运行 static gates、lint、coverage、snapshot replay 和 artifact gates,Node 26 兼容性 job 运行 `pnpm run check:node-compat`。各 lane 调度器并发运行来自 `package.json` 的独立门禁:constraints、typecheck、lint、coverage、snapshot replay、`doc-sync` 成员、module graph 新鲜度、`knip` 和 echo-agent 冒烟测试。 +keyless GitHub 工作流有八个 job:五个 Node 24 lane 分别运行 static gates、lint、coverage、snapshot replay 和 artifact gates,三个兼容性 job 在 Node 22.19、24 和 26 上运行 `pnpm run check:node-compat`。各 lane 调度器并发运行来自 `package.json` 的独立门禁:constraints、typecheck、lint、coverage、snapshot replay、`doc-sync` 成员、module graph 新鲜度、`knip` 和 echo-agent 冒烟测试。 `pnpm run build` 供给 artifact lane,`publint`、`verify-node-next-types` 和 built-bin 冒烟测试等待 build 输出。单独的真实 API 工作流带密钥运行 `pnpm run test:e2e`,并设置 `DSH_E2E_MAX_WORKERS=14`。 diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index 6b15512503..483eb9de2a 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -144,7 +144,7 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; | [Generated persistence log event catalog](implemented/process/2026-07-04-persistence-log-catalog.md) | 2026-07-04 | | [One gated in-file format for RFCs](implemented/process/2026-07-05-uniform-rfc-format.md) | 2026-07-05 | | [Generated plugin config catalog](implemented/process/2026-07-06-generated-config-catalog.md) | 2026-07-06 | -| [Lower the Node engines floor to 22.18](implemented/process/2026-07-06-node-22-18-floor.md) | 2026-07-06 | +| [Raise the Node LTS engine floor to 22.19](implemented/process/2026-07-06-node-engine-floor.md) | 2026-07-06 | | [Parallel GitHub CI gates](implemented/process/2026-07-06-parallel-github-ci-gates.md) | 2026-07-06 | | [Parallel pre-push gates](implemented/process/2026-07-06-parallel-pre-push-gates.md) | 2026-07-06 | diff --git a/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md b/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md index 74b68ca2f3..660b9821ff 100644 --- a/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md +++ b/docs/rfc/implemented/architecture/2026-06-24-web-capability-seam.md @@ -66,7 +66,7 @@ flowchart LR `@deepseek-ai/dsh-web` depends only on Cordis and low-level harness support. It declares `ctx.web`, provider interfaces, request/result types, the provider status type, and error codes. It does not import tool, agent, session, LLM, or provider packages. -Provider packages depend on `@deepseek-ai/dsh-web` and Cordis. They own credentials, endpoint config, provider-specific request mapping, provider-specific response parsing, and provider-specific error translation into `WebError`. They issue network requests with the platform-native `fetch` (Node 22.18), mirroring `@deepseek-ai/dsh-llm-deepseek`'s adapter, NOT a cordis HTTP-client service (`ctx.http`/`@cordisjs/plugin-http`) — even where a Perplexity provider's request is shaped like an OpenAI-compatible chat completion, that wire shape is a provider-private detail and does not make the provider depend on `ctx.llm`. A provider does NOT own the `ctx.web` key (two search providers cannot both own it): like `dsh-llm-deepseek`, each provider package is a function/namespace plugin (`inject: ['web']`) whose `apply` constructs the backend and calls `ctx.web.registerSearchProvider` / `registerFetchProvider`. `@deepseek-ai/dsh-web` is the `export default` service that owns the key. +Provider packages depend on `@deepseek-ai/dsh-web` and Cordis. They own credentials, endpoint config, provider-specific request mapping, provider-specific response parsing, and provider-specific error translation into `WebError`. They issue network requests with platform-native `fetch` at the repo's Node floor, mirroring `@deepseek-ai/dsh-llm-deepseek`'s adapter, NOT a cordis HTTP-client service (`ctx.http`/`@cordisjs/plugin-http`) — even where a Perplexity provider's request is shaped like an OpenAI-compatible chat completion, that wire shape is a provider-private detail and does not make the provider depend on `ctx.llm`. A provider does NOT own the `ctx.web` key (two search providers cannot both own it): like `dsh-llm-deepseek`, each provider package is a function/namespace plugin (`inject: ['web']`) whose `apply` constructs the backend and calls `ctx.web.registerSearchProvider` / `registerFetchProvider`. `@deepseek-ai/dsh-web` is the `export default` service that owns the key. `@deepseek-ai/dsh-tool-web` depends on `@deepseek-ai/dsh-web`, `@deepseek-ai/dsh-tools`, `@deepseek-ai/dsh-system-prompt`, and Cordis. It never imports concrete provider packages. diff --git a/docs/rfc/implemented/process/2026-06-11-quality-gates.md b/docs/rfc/implemented/process/2026-06-11-quality-gates.md index 52823b8222..505cea90ff 100644 --- a/docs/rfc/implemented/process/2026-06-11-quality-gates.md +++ b/docs/rfc/implemented/process/2026-06-11-quality-gates.md @@ -14,7 +14,7 @@ Every AGENTS.md promise gets a command that exits non-zero, wired into git hooks - ESLint strict-type-checked + @stylistic (the house style, enforced); vendored code excluded. - Per-file 100% coverage on `packages/*/src` (v8); unreachable defensive guards carry `/* v8 ignore */ ` with stated reasons instead of deletion. - knip (dead code/deps), publint (package correctness), workspace constraints (workspace rules: private, cordis peer+dev, uniform version, ESM), and a NodeNext consumer typecheck for built package declarations. -- lefthook pre-commit (lint staged, typecheck, vendor-manifest guard) and pre-push (tests, hygiene); CI runs the full matrix on node 22.18/24/26 plus a demo smoke test driving the echo-agent end to end. +- lefthook pre-commit (lint staged, typecheck, vendor-manifest guard) and pre-push (tests, hygiene); CI runs the full matrix on node 22.19/24/26 plus a demo smoke test driving the echo-agent end to end. ## Consequences diff --git a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md b/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md deleted file mode 100644 index 8e0c736fd9..0000000000 --- a/docs/rfc/implemented/process/2026-07-06-node-22-18-floor.md +++ /dev/null @@ -1,34 +0,0 @@ -# RFC: Lower the Node engines floor to 22.18 - -Status: implemented - -## Problem - -The root `engines.node` was `>=24`, which excluded the entire Node 22 LTS line for no runtime reason. The harness has exactly two Node features whose availability gates the floor, and both are satisfied well below Node 24 — so the floor was higher than the code actually requires. Pinning it honestly widens the supported install base (Node 22 LTS is in service until 2027) without weakening any guarantee, provided CI proves the claim on the floor version rather than merely asserting it in a manifest. - -## Decision - -Set `engines.node` to `^22.18.0 || >=24.0.0` (Node 22.18+ on the LTS line, or 24+) and test it on the keyless CI matrix `['22.18', 24, 26]`. The real-API e2e workflow stays on Node 24 because it exercises API integration rather than the runtime floor. Two Node features gate the range, each with its own LTS-line and Current-line unflag point: - -- **`node:sqlite`** — `packages/session-persistence/session-persistence-sqlite` does a top-level `import { DatabaseSync } from 'node:sqlite'`. The module dropped its `--experimental-sqlite` flag requirement at **22.13** (LTS) and **23.4** (Current); before those, importing it throws at load. -- **Native TypeScript type-stripping** — the `packages/ui/stdio-agent/tests/built-bin.e2e.ts` smoke boots the published `lib/bin.js` under plain `node` (no tsx) and loads the example's `.ts` plugins (`mock-llm.ts`, `echo-tool.ts`). Type-stripping is the default from **22.18** (LTS) and **23.6** (Current); before those it needs `--experimental-strip-types`. - -On the 22.x line both features clear at **22.18** (the later of 22.13/22.18), so `^22.18.0` is the LTS floor. The range is **disjoint** rather than an open `>=22.18` because the Node **23.0–23.5** window still has at least one feature flagged (sqlite until 23.4, stripping until 23.6): `>=22.18` would advertise support there, where the sqlite backend throws `ERR_UNKNOWN_BUILTIN_MODULE` at load. Node 23 is non-LTS and already end-of-life, so rather than carve out `>=23.6` the range skips the whole line and resumes at `>=24.0.0` — the same shape several of the repo's own dependencies already declare (`^22.18.0 || >=24.11.0`). - -`@types/node` is pinned to the 22.x line (`^22.20.0`) to match the floor: reaching for a Node 23+/24+/25+ API then fails `tsc` on every machine and in the typecheck gate, rather than compiling clean and surviving to a runtime failure only the 22.18 matrix leg could catch. The whole tree typechecks clean against the Node 22 type surface today, so the pin costs nothing. - -## Consequences - -- The supported base widens to the Node 22 LTS line, and the `['22.18', 24, 26]` matrix proves it on every push and PR rather than trusting the manifest. -- The built-bin smoke needs no version-conditional flag: at 22.18 type-stripping is already the default, so the test stays the plain `node lib/bin.js` path it documents. -- A future change reaching for a Node 23+ API fails `tsc` immediately (the `@types/node` pin); one reaching for an API added in 22.19/22.20 — inside the 22.x type surface but above the floor — is caught instead by the 22.18 matrix leg. Either way the floor must move in the same change. -- The `vendor/hmr` and `vendor/loader` comments about Node 24 module-cache internals are unaffected — they describe dev-time HMR loader behavior, not the shipped runtime contract, and are pinned vendored source. - -## Alternatives considered - -- **Floor `>=22.13` (the `node:sqlite` boundary) plus `--experimental-strip-types` in the built-bin smoke on 22.13–22.17.** Rejected: it adds a version-conditional test flag for one narrow range and dresses up an experimental-flag dependency as first-class support. 22.18 clears both boundaries with zero test special-casing, and the five-patch gap below it buys nothing real. -- **Keep `>=24`.** Rejected: it excludes Node 22 LTS with no runtime justification once the two boundaries above are known. -- **Open-ended `>=22.18`.** Rejected: it advertises support for Node 23.0–23.5, where `node:sqlite` (until 23.4) or type-stripping (until 23.6) is still flagged, so the sqlite backend throws at load. The disjoint `^22.18.0 || >=24.0.0` matches the real runtime boundary. -- **Include Node 23.6+ (`^22.18.0 || >=23.6.0`).** Rejected: 23.6+ does run both features unflagged, but Node 23 is end-of-life — advertising a dead release line adds a range term (and, to back it, a CI leg) for a runtime no deployment should use. 24 is the meaningful resumption point, and the 22.18 and 24 legs already bracket the same unflagged code paths. -- **Matrix `[22, 24, 26]` (latest 22.x) instead of pinning `22.18`.** Rejected: "latest 22.x" drifts upward over time and would silently stop exercising the declared floor. Pinning the floor version is what makes the matrix a proof of the claim rather than a proof of some newer 22.x. -- **Keep `@types/node` ahead of the floor (`^25`).** Rejected: types ahead of the runtime floor let a Node 24/25-only API compile clean and fail only at runtime on 22.18 — exactly the "green types, broken product" gap. Pinning `@types/node` to the 22.x line turns that into a compile error everywhere, and the tree already typechecks clean against the Node 22 surface, so the pin is free. diff --git a/docs/rfc/implemented/process/2026-07-06-node-engine-floor.md b/docs/rfc/implemented/process/2026-07-06-node-engine-floor.md new file mode 100644 index 0000000000..72c9f09eda --- /dev/null +++ b/docs/rfc/implemented/process/2026-07-06-node-engine-floor.md @@ -0,0 +1,37 @@ +# RFC: Raise the Node LTS engine floor to 22.19 + +Status: implemented + +## Problem + +The Node 22 branch of the root `engines.node` range is a contract for the installed workspace, not only for the runtime APIs the harness source calls directly. It must be no lower than package `engines.node` declarations for dependencies the workspace installs on that branch; otherwise `pnpm install --engine-strict` fails at an advertised LTS version, and non-strict installs run outside a dependency's supported runtime. + +## Decision + +Set `engines.node` to `^22.19.0 || >=24.0.0` and test the keyless CI compatibility matrix on `['22.19', 24, 26]`. The real-API e2e workflow stays on Node 24 because it exercises API integration rather than the runtime floor. + +Two Node features gate the source runtime: + +- **`node:sqlite`** — `packages/session-persistence/session-persistence-sqlite` does a top-level `import { DatabaseSync } from 'node:sqlite'`. The module dropped its `--experimental-sqlite` flag requirement at **22.13** (LTS) and **23.4** (Current); before those, importing it throws at load. +- **Native TypeScript type-stripping** — the `packages/ui/stdio-agent/tests/built-bin.e2e.ts` smoke boots the published `lib/bin.js` under plain `node` (no tsx) and loads the example's `.ts` plugins (`mock-llm.ts`, `echo-tool.ts`). Type-stripping is the default from **22.18** (LTS) and **23.6** (Current); before those it needs `--experimental-strip-types`. + +Those source features clear on the 22.x line at **22.18**, but the installed Pi adapter dependency raises the advertised LTS floor. `@deepseek-ai/dsh-llm-pi-ai` depends on `@earendil-works/pi-ai@0.79.3`, whose package declares `engines.node >=22.19.0`, so the LTS floor is **22.19**. The 24.x branch remains `>=24.0.0`. The disjoint range excludes Node 23 entirely: Node 23.0–23.5 still has at least one flagged source feature, and the 23 line is non-LTS/EOL, so advertising `>=23.6` would add a dead release line and a CI leg no deployment should use. + +`@types/node` remains pinned to the 22.x line (`^22.20.0`) to match the LTS support line: reaching for a Node 23+/24+/25+ API fails `tsc` on every machine and in the typecheck gate, rather than compiling clean and surviving to a runtime failure only a floor matrix leg could catch. The whole tree typechecks clean against the Node 22 type surface today, so the pin costs nothing. + +## Consequences + +- The advertised LTS branch no longer undercuts the Pi adapter dependency floor. +- CI proves the Node 22 LTS floor directly with Node 22.19, keeps the Node 24 branch on `node: 24`, and keeps Node 26 for the next even line. +- The built-bin smoke needs no version-conditional flag: at 22.19 type-stripping is already the default, so the test stays the plain `node lib/bin.js` path it documents. +- A future dependency or source API that raises the runtime floor must move `engines.node`, the compatibility matrix, and this RFC in the same change. + +## Alternatives considered + +- **Keep `^22.18.0 || >=24.0.0`.** Rejected: it advertises an LTS version lower than the Pi adapter dependency floor. `@earendil-works/pi-ai@0.79.3` requires `>=22.19.0`. +- **Downgrade or pin `@earendil-works/pi-ai` to preserve the 22.18 advertised range.** Rejected: the current Pi adapter dependency is part of the intended workspace, and 22.19 is still inside the Node 22 LTS line. +- **Floor `>=22.13` (the `node:sqlite` boundary) plus `--experimental-strip-types` in the built-bin smoke on 22.13–22.17.** Rejected: it adds a version-conditional test flag for one narrow range and dresses up an experimental-flag dependency as first-class support. The Pi adapter dependency already requires a higher LTS floor. +- **Open-ended `>=22.19`.** Rejected: it advertises support for Node 23.0–23.5, where `node:sqlite` (until 23.4) or type-stripping (until 23.6) is still flagged. +- **Include Node 23.6+ (`^22.19.0 || >=23.6.0`).** Rejected: 23.6+ does run both source features unflagged, but Node 23 is end-of-life; advertising a dead release line adds a range term and a CI leg for a runtime no deployment should use. +- **Matrix `[22, 24, 26]` instead of pinning `22.19`.** Rejected: floating major-version entries drift upward over time and silently stop exercising the declared LTS floor. +- **Keep `@types/node` ahead of the floor (`^25`).** Rejected: types ahead of the runtime floor let a Node 24/25-only API compile clean and fail only at runtime on 22.x. Pinning `@types/node` to the 22.x line turns that into a compile error everywhere. diff --git a/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md b/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md index 1b348b1515..e567c4ff24 100644 --- a/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md +++ b/docs/rfc/implemented/testing/2026-06-19-real-api-e2e-ci.md @@ -54,7 +54,7 @@ The repo secret is named `DEEPSEEK_API_KEY_EXTERNAL`; it is mapped to the `DEEPS ### Scope, runtime shape -Run **only** `test:e2e`. The keyless gates (typecheck/lint/coverage/snapshot/build/hygiene) already run in ci.yml on every push and PR; repeating them here would duplicate signal and slow the real-API job. No build step — e2e tests run unbuilt via tsx + the tsconfig paths map. Single Node 24 (the primary line): these tests exercise API integration, not node-version compatibility, which ci.yml's Node 22.18/24/26 matrix owns. `vitest.e2e.config.ts` runs files through a bounded worker pool (`DSH_E2E_MAX_WORKERS`, default `4`, CI value `14`) so CI and local with-key runs parallelize independent files while retaining a one-line serial escape hatch for quota investigations. `timeout-minutes: 45` bounds a wedged run given 120s/test and `retry: 2`. `cancel-in-progress` is enabled only for `pull_request` runs — a superseded PR run is on a stale commit and worth cancelling, whereas a push/schedule run is already producing the post-merge/nightly signal and is never cancelled. +Run **only** `test:e2e`. The keyless gates (typecheck/lint/coverage/snapshot/build/hygiene) already run in ci.yml on every push and PR; repeating them here would duplicate signal and slow the real-API job. No build step — e2e tests run unbuilt via tsx + the tsconfig paths map. Single Node 24 (the primary line): these tests exercise API integration, not node-version compatibility, which ci.yml's Node 22.19/24/26 matrix owns. `vitest.e2e.config.ts` runs files through a bounded worker pool (`DSH_E2E_MAX_WORKERS`, default `4`, CI value `14`) so CI and local with-key runs parallelize independent files while retaining a one-line serial escape hatch for quota investigations. `timeout-minutes: 45` bounds a wedged run given 120s/test and `retry: 2`. `cancel-in-progress` is enabled only for `pull_request` runs — a superseded PR run is on a stale commit and worth cancelling, whereas a push/schedule run is already producing the post-merge/nightly signal and is never cancelled. ## Security diff --git a/package.json b/package.json index 60770f0d1f..6710b6bf44 100644 --- a/package.json +++ b/package.json @@ -5,7 +5,7 @@ "type": "module", "packageManager": "pnpm@11.7.0", "engines": { - "node": "^22.18.0 || >=24.0.0" + "node": "^22.19.0 || >=24.0.0" }, "workspaces": [ "vendor/*", diff --git a/packages/session-persistence/session-persistence-sqlite/README.md b/packages/session-persistence/session-persistence-sqlite/README.md index f3601140d2..c14d3a70ea 100644 --- a/packages/session-persistence/session-persistence-sqlite/README.md +++ b/packages/session-persistence/session-persistence-sqlite/README.md @@ -8,7 +8,7 @@ A SQLite durable session-persistence backend — a second `SessionPersistence` i Each `SessionEvent` maps 1:1 onto a row in an `events` table `(session_id, seq, type, time, data, source_event_seqs, surface_op)` — `data` is the event payload as JSON text, so the row shape is the event verbatim (including `assistant/chunk`, keeping `seq` contiguous). The two `TEXT` columns `source_event_seqs` and `surface_op` are nullable; they store the event's optional surface-metadata fields (see [session surface](../../../docs/rfc/implemented/architecture/2026-06-18-session-surface.md)). Out-of-log metadata (`SessionHeader`) lives in a `sessions` row. A `sessions` row is written only by the first `append` — its existence is the lazy-materialization signal (`list` reports exactly the sessions that have a row), so no separate column is needed. -The repo's `engines.node` is `^22.18.0 || >=24.0.0` (Node 22.18+ or 24+). `node:sqlite` ships without the `--experimental-sqlite` flag from Node 22.13 (LTS) and 23.4 / 24 (Current) on; the range deliberately excludes the Node 23.0–23.3 window, where the module is still flagged and this backend's top-level `import { DatabaseSync } from 'node:sqlite'` would throw at load. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software). +The repo's `engines.node` is `^22.19.0 || >=24.0.0` (Node 22.19+ or 24+), matching the LTS floor required by the installed Pi adapter dependency; `node:sqlite` itself ships without the `--experimental-sqlite` flag from Node 22.13 (LTS) and 23.4 / 24 (Current) on. The range deliberately excludes Node 23 because that line is non-LTS/EOL and still has flagged runtime features before 23.6. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software). ## Contract semantics over rows diff --git a/packages/ui/stdio-agent/tests/built-bin.e2e.ts b/packages/ui/stdio-agent/tests/built-bin.e2e.ts index 38e0170ff5..7b84fad65b 100644 --- a/packages/ui/stdio-agent/tests/built-bin.e2e.ts +++ b/packages/ui/stdio-agent/tests/built-bin.e2e.ts @@ -77,7 +77,7 @@ async function makeConsumer(welcome: string, disabledBrokenEntry = false): Promi await symlink(abs, target) } // The example's mock model + echo tool are example-local TS plugins (Node - // 22.18+ — the engines floor — strips types natively, so plain `node` loads + // 22.19+ — the engines floor — strips types natively, so plain `node` loads // them); they import the workspace packages the symlinked node_modules now // provides. await cp(join(repoRoot, 'examples/echo-agent/src'), join(dir, 'src'), { recursive: true }) diff --git a/packages/web/web-fetch-local/src/provider.ts b/packages/web/web-fetch-local/src/provider.ts index acfca02e78..2175a62770 100644 --- a/packages/web/web-fetch-local/src/provider.ts +++ b/packages/web/web-fetch-local/src/provider.ts @@ -1,6 +1,6 @@ /** * `LocalFetchProvider`: a `WebFetchProvider` that retrieves a concrete public - * HTTP(S) URL with the platform-native `fetch` (Node 22.18) and returns a status + * HTTP(S) URL with platform-native `fetch` at the repo's Node floor and returns a status * code plus bounded decoded content. It owns SAFE RESOURCE RETRIEVAL — URL * validation, redirect policy, timeout, abort, byte caps, charset decoding, * content-type classification, binary rejection — but NOT presentation diff --git a/packages/web/web-search-deepseek/src/provider.ts b/packages/web/web-search-deepseek/src/provider.ts index ade27721a4..4c9679eb8e 100644 --- a/packages/web/web-search-deepseek/src/provider.ts +++ b/packages/web/web-search-deepseek/src/provider.ts @@ -12,7 +12,7 @@ * `web_search_tool_result` block (native search did not trigger), it throws * `WEB_PROVIDER_ERROR` rather than degrading to prose-scraping. * - * Network requests use platform-native `fetch` (Node 22.18), mirroring + * Network requests use platform-native `fetch` at the repo's Node floor, mirroring * `@deepseek-ai/dsh-llm-deepseek`'s adapter — not a cordis HTTP-client service. * The Anthropic wire shape is a provider-private detail and does NOT make this * provider depend on `ctx.llm`. diff --git a/packages/web/web-search-exa/src/provider.ts b/packages/web/web-search-exa/src/provider.ts index 8e3f7d8b02..6a764fae93 100644 --- a/packages/web/web-search-exa/src/provider.ts +++ b/packages/web/web-search-exa/src/provider.ts @@ -6,7 +6,7 @@ * `title`, the first highlight as `snippet`, and `publishedDate` as * `publishedAt`. * - * Network requests use platform-native `fetch` (Node 22.18), mirroring + * Network requests use platform-native `fetch` at the repo's Node floor, mirroring * `@deepseek-ai/dsh-llm-deepseek`'s adapter — not a cordis HTTP-client service. * * @module @deepseek-ai/dsh-web-search-exa/provider diff --git a/packages/web/web-search-perplexity/src/provider.ts b/packages/web/web-search-perplexity/src/provider.ts index f4a12fb415..506a03be59 100644 --- a/packages/web/web-search-perplexity/src/provider.ts +++ b/packages/web/web-search-perplexity/src/provider.ts @@ -5,7 +5,7 @@ * structured `search_results[]` for `sources[]`, falling back to the URL-only * `citations[]` when `search_results` is absent. * - * Network requests use platform-native `fetch` (Node 22.18), mirroring + * Network requests use platform-native `fetch` at the repo's Node floor, mirroring * `@deepseek-ai/dsh-llm-deepseek`'s adapter. The OpenAI-compatible request shape * is a provider-private detail and does NOT make this provider depend on * `ctx.llm`. From 46a719bba6f7e1e3f75e997f4f804fba28e55b2c Mon Sep 17 00:00:00 2001 From: pku-xht <170163488+pku-xht@users.noreply.github.com> Date: Tue, 7 Jul 2026 09:40:07 +0000 Subject: [PATCH 11/24] docs(rfc): propose Claude Code and Codex subagent backends Out-of-process delegation to external coding agents as two new subagent seam backends, exposed as subagent_claude_code / subagent_codex tools. Verified against @anthropic-ai/claude-agent-sdk 0.3.202 and codex CLI 0.142.5 via keyless spikes; includes the dsh-subagent-process extraction plan, isolation/permission stances, and tiered test coverage. --- docs/rfc/INDEX.md | 1 + ...claude-code-and-codex-subagent-backends.md | 87 +++++++++++++++++++ 2 files changed, 88 insertions(+) create mode 100644 docs/rfc/proposed/feature/2026-07-07-claude-code-and-codex-subagent-backends.md diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index 5e23a876fb..c692c6ffea 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -12,6 +12,7 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; | [Multiplex concurrent ACP sessions over one connection](proposed/feature/2026-06-14-acp-multi-session.md) | 2026-06-14 | | [Optional Code Mode — model writes TypeScript against an SDK of all tools](proposed/feature/2026-06-15-optional-code-mode.md) | 2026-06-15 | | [Pre-tool input rewrite — a consistent design](proposed/feature/2026-06-30-pre-tool-input-rewrite.md) | 2026-06-30 | +| [Claude Code and Codex subagent backends (out-of-process delegation to external coding agents)](proposed/feature/2026-07-07-claude-code-and-codex-subagent-backends.md) | 2026-07-07 | ### Simplification diff --git a/docs/rfc/proposed/feature/2026-07-07-claude-code-and-codex-subagent-backends.md b/docs/rfc/proposed/feature/2026-07-07-claude-code-and-codex-subagent-backends.md new file mode 100644 index 0000000000..3de3c7d9be --- /dev/null +++ b/docs/rfc/proposed/feature/2026-07-07-claude-code-and-codex-subagent-backends.md @@ -0,0 +1,87 @@ +# RFC: Claude Code and Codex subagent backends (out-of-process delegation to external coding agents) + +Status: proposed + +## Problem + +The subagent seam ([the seam RFC](../../implemented/feature/2026-06-21-subagent-capability-seam.md)) hosts multiple named providers on `ctx.subagents`, and the ACP backend ([the ACP backend RFC](../../implemented/feature/2026-06-22-acp-subagent-backend.md)) proved the seam generalizes across a process boundary; its Future-providers section explicitly named the Codex app-server and the Claude Code Agent SDK as mechanically similar siblings. Those two are the engines actually worth delegating to today: a harness turn should be able to hand a self-contained task to a real Claude Code or a real Codex — a separate product with its own model, tools, and sandbox — and get back one final answer, without the parent deployment leaking its secrets into the child or the child's behavior silently depending on whatever `~/.claude` / `~/.codex` state exists on the host machine. + +## Proposal + +Two sibling provider packages, structural variants of the ACP backend, plus one extraction: + +- `@deepseek-ai/dsh-subagent-claude-code` — drives a Claude Code child through `@anthropic-ai/claude-agent-sdk`'s `query()` (the SDK runs in the parent process and spawns its bundled `claude` CLI as the subprocess). Provider name `claude-code`: the child is the Claude Code *product*, not an Anthropic model adapter — "claude" stays reserved for a future `dsh-llm` adapter. +- `@deepseek-ai/dsh-subagent-codex` — spawns `codex app-server` and drives one thread/turn over its JSON-RPC-over-stdio protocol with a hand-rolled newline-JSON client (~200–300 lines) in the package. +- `@deepseek-ai/dsh-subagent-process` — a pure library (the `subagent-inprocess` precedent) extracting what `dsh-subagent-acp` already carries and both new backends need: the credential env scrub (`SENSITIVE_ENV_PATTERN`/`buildChildEnv`), the EOF → SIGTERM → SIGKILL dispose ladder, and new isolated-config-dir helpers (`mkdtemp` create, best-effort remove). The ACP backend migrates onto it; `bash-local`'s sibling copy is left alone to bound the change. + +Both providers copy the ACP backend's seam posture verbatim: fresh child per `start`, exactly one prompt round-trip, capabilities all `false`, `inheritsParentContext: false`, `request.parent`/`request.agentOptions` ignored, `id = AgentId(randomUUID())`, `result` never rejects — child-level failure flattens to a stop reason and the original error goes to `ctx.logger` via an `onError` spec callback. Model exposure is zero new code: `dsh-tool-subagent` is loaded once per provider with a distinct `toolName` (`subagent_claude_code`, `subagent_codex`). No new session events are needed — the only model-visible artifact is the tool result, so reconstructability holds exactly as it did for ACP. To be explicit about the boundary: the session log reconstructs the model-visible transcript, not workspace mutation history — a child granted write access mutates files as an ambient side effect outside the log, exactly as the bash tools and the ACP backend already do; replay reproduces requests, not the disk. + +## Verified interface facts (pinned versions) + +Both integration surfaces were verified against pinned implementations before this proposal — types and bundled source read, keyless spikes run — not from vendor docs alone. The pins are the verification baseline, not a runtime contract: the backends perform no runtime version probe (no `codex --version` gate, no SDK version sniffing). Compatibility is enforced at development time — every dependency bump re-runs the keyless suites against the real load path — and at runtime by failing loudly: a protocol-level surprise settles `error` via `onError`, never a silent misbehavior. + +**`@anthropic-ai/claude-agent-sdk` 0.3.202.** `options.env` REPLACES the child environment (no merge with `process.env`), which is exactly what the scrub needs. `settingSources` defaults to loading ALL filesystem settings — isolation requires explicitly passing `[]`. Result subtypes are `success` | `error_during_execution` | `error_max_turns` | `error_max_budget_usd` | `error_max_structured_output_retries`. On abort the SDK escalates the CLI child itself: stdin EOF immediately, SIGTERM ~2s later if the child ignores it (observed; no leftover processes) — no bespoke kill fallback needed. `outputFormat: {type: 'json_schema'}` and an `agents` option exist, giving future landing points for the seam's `outputSchema` capability and named subagent types; both are out of scope here. + +**codex CLI 0.142.5, `codex app-server` (v2 vocabulary).** LF-delimited JSON, JSON-RPC 2.0 shapes with the `"jsonrpc"` header omitted. + +- Lifecycle: `initialize{clientInfo}` + `initialized` → `thread/start` (accepts `cwd`, `model`, `sandbox`, `approvalPolicy`, `ephemeral`; succeeds unauthenticated) → `turn/start{threadId, input:[{type:'text',text}]}` returns an `inProgress` turn immediately; the terminal signal is the `turn/completed` notification carrying `Turn{status: completed|interrupted|failed|inProgress, error}`. +- Approvals are server-initiated requests — `item/commandExecution/requestApproval`, `item/fileChange/requestApproval`, `item/permissions/requestApproval`, `item/tool/requestUserInput`, `mcpServer/elicitation/request` — answered with `accept`/`decline`-family decisions. +- Auth: `account/login/start{type:'apiKey', apiKey}` is a first-class RPC and `account/read` reports `requiresOpenaiAuth` — and an unauthenticated `turn/start` does NOT fail fast (it hangs in retry), so the backend MUST pre-check auth and settle `error` loudly instead of waiting on the turn. +- Isolation: `CODEX_HOME` redirection is honored (the `initialize` response echoes it, so tests can assert isolation), and `ephemeral: true` threads leave no session files at all. + +## Isolation and credentials + +Deployments authenticate with API keys only, and the child must not see the host user's Claude Code / Codex configuration: behavior has to be a function of `cordis.yml` alone. Each run gets a fresh `mkdtemp` config dir — `CLAUDE_CONFIG_DIR` for Claude Code (paired with an explicit `settingSources: []`), `CODEX_HOME` for Codex — removed best-effort on dispose; a config field can pin a persistent dir instead. The child env reuses the ACP backend's `buildChildEnv` semantics verbatim via the extraction: the ambient env is forwarded MINUS credential-shaped vars (`/KEY|SECRET|TOKEN/i`), with `config.env` layered on top — so `PATH`, `HOME`, `TMPDIR`, locale, and proxy vars survive and the CLIs run normally, while only credential-shaped ambient vars are scrubbed (`ANTHROPIC_API_KEY` enters explicitly through `config.env` for Claude Code), and the Codex key travels via the `account/login/start` RPC into the isolated `CODEX_HOME` rather than a hand-written `auth.json`. + +## Permission and approval policy + +Instead of collapsing to ACP's single `permission: allow|reject` knob, each backend exposes its engine's native vocabulary as config, with conservative defaults: Claude Code gets `permissionMode` (default `default`) plus `permission: allow|reject` (default `reject`) as the `canUseTool` auto-answer for whatever falls through; Codex gets `sandboxMode` (default `read-only`) and `approvalPolicy` (default `never`) plus the same `permission` fallback for approval requests that still arrive. Defaults are deliberately do-no-harm (the out-of-box child cannot write files); examples demonstrate opening up (`acceptEdits` / `workspace-write`). The mechanical rule: EVERY server-initiated request is settled programmatically and promptly — the enumerated approval/user-input/elicitation requests by the configured policy, an unknown request method with a JSON-RPC method-not-found error response (never left pending), unknown notifications consumed — so no child request can wedge a turn waiting on an answer that will never come. Prompts never reach a human in this cut, matching ACP. + +## StopReason mapping + +Claude Code: `success` → `completed`; `error_max_turns`, `error_during_execution`, `error_max_budget_usd`, `error_max_structured_output_retries` → `error` (aligning with the ACP call on `max_turn_requests`: an unfinished task is not success); generator abort → `aborted`; anything unknown → `error`. Codex: `Turn.status` `completed` → `completed`; `interrupted` → `aborted`; `failed` with `codexErrorInfo: 'contextWindowExceeded'` → `max-tokens`, any other `failed` → `error`; transport/spawn/auth-precheck failure → `error` (or `aborted` if cancel was requested). In both, `cancel()` is the ACP shape: flag + abort/interrupt + a cancel-settled race arm so an uncooperative child cannot stall the result. + +Liveness posture, stated explicitly: teardown timing is config, turn duration is not. Both backends take the dispose ladder's grace periods as defaulted validated config fields (the ACP backend's `disposeEofGraceMs`/`disposeGraceMs` shape, carried by the extraction), but there is deliberately NO turn-duration or startup timeout — matching ACP, liveness during a turn belongs to the caller via `cancel()`/the abort signal, a subagent turn is legitimately minutes long, and the Codex auth precheck removes the one verified guaranteed-hang; a deployment wanting a wall-clock bound cancels from the parent. + +## Testing + +Named at every tier per the root AGENTS.md rule, and de-risked up front: + +- **Keyless unit/integration**, mirroring the ACP spec list per backend (round-trip and output accumulation, every stop mapping, both cancel paths, already-aborted, permission auto-answer under both policies, unknown-message tolerance, bad-command spawn failure, HMR provider cleanup, export shape, isolation assertions on child env and temp-dir removal; Codex adds the auth-precheck failure path). Claude Code's harness is a scripted fake `claude` executable behind `pathToClaudeCodeExecutable` driven by the REAL SDK — a spike already passed end-to-end keyless in 24ms (the fake CLI answers one `control_request/initialize` and speaks plain stream-json, ~40 lines). Codex's harness is a scripted mock app-server subprocess speaking the verified wire protocol, the `mock-acp-server.ts` shape. +- **With-key e2e** per backend: the real engine does real file work verified on disk, under a pinned opened-up config so acceptance and the do-no-harm defaults don't collide — `permissionMode: 'acceptEdits'` for Claude Code, `sandboxMode: 'workspace-write'` + `approvalPolicy: 'never'` for Codex; self-skips report exactly what is missing (binary vs key). CI has no secrets, so these run locally per the with-key policy. +- **Snapshot**: deferred as `TODO(claude-code-subagent-replay)` / `TODO(codex-subagent-replay)` — the same distinct replay shape the ACP backend deferred ([the per-session replay RFC](../../implemented/testing/2026-06-22-subagent-snapshot-replay.md)); the keyless suites carry deterministic coverage meanwhile. + +## Alternatives considered + +### Why not the official `@openai/codex-sdk` instead of a hand-rolled client? + +The dispose ladder and env scrub require owning the child process (spawn args, env, signals, exit await); the SDK hides the process. The wire format is trivial to frame (LF JSON), the shapes are generatable per pinned version (`codex app-server generate-json-schema`), and the repo precedent (`hook-protocol`) is to own thin protocol cores rather than wrap someone's runtime. The SDK would save protocol-evolution maintenance but costs the exact control this backend exists to have. + +### Why not a model-visible `subagent_type` parameter (one Task-style tool)? + +Claude Code's own Task tool puts the subagent type in the model-facing schema, selecting a prompt-plus-toolset persona. Here the choice is between EXECUTION ENGINES, and only the deployer knows which engines have credentials configured — so selection stays deployment config, preserving `dsh-tool-subagent`'s documented one-provider-per-tool contract. A persona-style type selector would be a separate RFC against the tool, not the backends. + +### Why not login-state credentials and the user's own config? + +Inheriting `~/.claude` / `~/.codex` (subscription login, user settings, skills, MCP servers) would make child behavior depend on host-machine state and punch an implicit exception through the "credentials enter explicitly via `config.env`, never ambiently" rule the ACP backend and bash executor established. API-key-only plus forced config-dir isolation keeps runs reproducible; deployments wanting shared state can point the config-dir field at a persistent directory deliberately. + +### Why not a driver-injection seam for the Claude Code keyless tests? + +Injecting a fake `query()` would mock our own boundary and leave the real SDK load path untested (the real-over-mock policy in docs/testing.md). The risk that justified considering it — the SDK↔CLI stream-json control protocol being internal — was retired by the spike: the fake-CLI harness works against the real pinned SDK today. If an SDK upgrade breaks the mock, the keyless suite fails the upgrade PR, which is the gate working. + +### Why not ACP adapters (e.g. `claude-code-acp`) reusing the existing backend? + +Community shims wrap both engines in ACP, which would make them "just config" on `dsh-subagent-acp`. But that inserts an unofficial third-party layer between the harness and the engine, erases the native control surfaces this RFC exposes (permissionMode, sandboxMode/approvalPolicy, config-dir isolation, apiKey RPC), and trades first-party protocol stability for a shim's release cadence. First-party surfaces — the Agent SDK and the app-server — are the supported integration points. + +## Acceptance criteria + +On a machine with both engines and keys configured: a REPL-driven model completes one real file task through `subagent_claude_code` and one through `subagent_codex`, the tool result being the child's final answer, with only `tool/call` + `tool/result` in the parent session log. Keyless suites pass at 100% per-file coverage in a credential-less environment, asserting isolation (scrubbed child env, no temp config dirs left after dispose) and that child behavior is unchanged by the presence or absence of `~/.claude` / `~/.codex`. Cancelling a parent turn quiesces both backends in bounded time with no leftover child processes. E2e suites self-skip cleanly, naming the missing prerequisite. + +## Risks + +- `codex app-server` is CLI-flagged experimental and its v1/v2 vocabularies coexist; the client pins 0.142.5, implements v2 only, and consumes unknown methods/notifications without crashing, but a future codex bump can still force rework (regenerate schemas and re-run the keyless suite on every bump — the development-time enforcement behind the no-runtime-version-probe stance above). +- The Claude Code fake-CLI mock rides an internal protocol: any SDK upgrade must go through the keyless suite, and a breaking control-protocol change means reworking the mock (fallback: the driver-injection seam rejected above becomes the escape hatch). +- The SDK's optionalDependencies weigh ~280MB per platform — accepted, and confined to the one backend package. +- The SDK's SIGKILL branch beyond EOF→SIGTERM was not observed and is trusted; e2e keeps a no-leftover-process assertion. +- Codex is a deployment prerequisite (no npm-bundled binary); a missing or incompatible binary surfaces as a loud spawn/protocol `error`, not a version probe. +- Every run pays a fresh child process and only the final answer surfaces — thoughts, tool cards, and usage are consumed and dropped; pooling, intermediate-progress surfacing, `sendMessage`/`resume`, `outputSchema` via the SDK's `outputFormat`, and named subagent types via the SDK's `agents` option are all deliberate deferrals. From 995ba1f1057de8769f158d1253cec112c960092c Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Tue, 7 Jul 2026 19:03:14 +0800 Subject: [PATCH 12/24] docs: tighten development onboarding wording --- docs/AGENTS.md | 2 +- docs/development.i18n.yaml | 4 ++-- docs/development.md | 4 ++-- docs/development.zh.md | 4 ++-- 4 files changed, 7 insertions(+), 7 deletions(-) diff --git a/docs/AGENTS.md b/docs/AGENTS.md index fed058f60a..771e30dcc6 100644 --- a/docs/AGENTS.md +++ b/docs/AGENTS.md @@ -16,7 +16,7 @@ Every fact has exactly one home — the tier whose job it is — and every other | [postmortem/](postmortem/README.md) | Incident stories — the only tier where war-story narrative belongs | — | | [cookbook/](cookbook/adding-a-package.md) | Step-by-step how-tos with numbered verify steps | Design rationale (→ the RFC each guide links) | | Package README | The per-package contract: config, semantics, limitations, extension points | JSDoc restatement, generated-catalog restatement (event/tool tables), other packages' concerns | -| [development.md](development.md) | Human-facing setup and daily workflow; a bilingual pair under the [i18n contract](i18n/README.md) | Gate-by-gate enumerations that drift from `package.json` scripts | +| [development.md](development.md) | First-stop contributor onboarding: local setup, daily workflow, and CI shape at summary level; a bilingual pair under the [i18n contract](i18n/README.md) | Runtime/version rationale (→ RFCs), gate-by-gate enumerations that drift from `package.json` scripts | | Generated catalogs: [cordis events](cordis-catalog/events.md), [cordis services](cordis-catalog/services.md), [tool-catalog](tool-catalog.md), [config-catalog](config-catalog.md), [persistence-catalog](persistence-catalog.md), [module-graph.md](module-graph.md) | Exhaustive enumerations regenerated from source, freshness-gated | Hand edits of any kind | | Skills (`.agents/skills/`) | Workflows: how to carry out a recurring task against the contracts | The contracts themselves (→ docs) | diff --git a/docs/development.i18n.yaml b/docs/development.i18n.yaml index c3926e69bd..06d0ff366c 100644 --- a/docs/development.i18n.yaml +++ b/docs/development.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -development.md: 3cb96968ccbe291c3cd94937c88cd414f18f2166 -development.zh.md: 98f3ad4cd12ecd73fd8e23ad48278d8bc23d2496 +development.md: 6d5bc28f412a888e239229305b983ffac08c737a +development.zh.md: eaa600d3a0a478d96848eb6866c06913a54fa428 diff --git a/docs/development.md b/docs/development.md index 3cb96968cc..6d5bc28f41 100644 --- a/docs/development.md +++ b/docs/development.md @@ -2,11 +2,11 @@ English | [中文](development.zh.md) -This guide covers the local setup needed to work on DeepSeek Harness and understand the local hooks, daily checks, and CI gates. +This onboarding guide helps project contributors get started with the local environment, daily workflow, and CI flow; see the RFCs for design rationale and technical trade-offs. ## Prerequisites -- Node.js `^22.19.0 || >=24.0.0` (22.19+ on the LTS line, or 24+). The LTS floor matches `@earendil-works/pi-ai`'s Node 22.19 dependency floor. The Node 23 line is excluded: `node:sqlite` (until 23.4) and native TS type-stripping (until 23.6) are still flagged there, and 23 is non-LTS/EOL. CI runs the compatibility matrix on Node 22.19, 24, and 26. +- Node.js supports 22.19+ and 24+. CI covers 22.19, 24, and 26; see the [Node engine floor RFC](rfc/implemented/process/2026-07-06-node-engine-floor.md). - Corepack-enabled pnpm. The repo pins `pnpm@11.7.0` in `package.json`; run `corepack enable` if `pnpm --version` does not resolve through Corepack. - Git. - Optional: a DeepSeek API key for the REPL/ACP agent demos and real-API e2e tests. diff --git a/docs/development.zh.md b/docs/development.zh.md index 98f3ad4cd1..eaa600d3a0 100644 --- a/docs/development.zh.md +++ b/docs/development.zh.md @@ -2,11 +2,11 @@ [English](development.md) | 中文 -本指南覆盖参与 DeepSeek Harness 开发所需的本地环境搭建,并帮助你理解本地钩子、日常检查与 CI 门禁。 +本文面向参与项目开发的贡献者,帮助你上手本地环境、日常工作流和 CI 流程。相关设计考量和技术取舍参见 RFC,不在这里展开。 ## 前置条件 -- Node.js `^22.19.0 || >=24.0.0`(即 LTS 线的 22.19+,或 24+)。LTS floor 匹配 `@earendil-works/pi-ai` 的 Node 22.19 依赖 floor。排除 Node 23 线:那里 `node:sqlite`(要到 23.4)和原生 TS 类型剥离(要到 23.6)仍需 flag,且 23 是非 LTS、已 EOL。CI 在 Node 22.19、24 和 26 上跑兼容性矩阵。 +- Node.js 支持 22.19+ 和 24+。CI 覆盖 22.19、24、26;见 [Node engine floor RFC](rfc/implemented/process/2026-07-06-node-engine-floor.md)。 - 启用了 Corepack 的 pnpm。仓库在 `package.json` 中钉住 `pnpm@11.7.0`;如果 `pnpm --version` 无法通过 Corepack 解析,先运行 `corepack enable`。 - Git。 - 可选:一个 DeepSeek API key,用于 REPL/ACP agent(智能体)演示和真实 API 的 e2e 测试。 From d1b52a063b75b89b655d14d5312306d8744948fc Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 21:06:14 +0800 Subject: [PATCH 13/24] fix review findings: own-property and plain-JSON discipline in the schema subset MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three Codex findings on json-schema.ts, one discipline: - required-declared and every value check now use Object.hasOwn — 'in' let inherited names (toString) satisfy required, dodge additionalProperties: false, and validate a declared property against the value's prototype member instead of a carried one - isObjectLike now means PLAIN JSON object (proto chain of at most one link, realm-agnostic): a Date annotation or a Map-as-properties no longer passes structurally and serializes lossily — they fail loud as subset violations - startInProcessRun asserts BEFORE the defensive structuredClone, so a hostile schema fails as OutputSchemaError, never a raw DataCloneError Also the type-equiv catalog gap: tools.md gains the structured-output subset vocabulary (4 blocks) with matching manifest entries. The driver index also drops the runtime internals from its public re-export (runs acquire it internally; no external consumer remains — see the following commit). --- docs/core-data-structures/tools.md | 34 ++ packages/core/tools/src/json-schema.ts | 35 +- packages/core/tools/tests/json-schema.spec.ts | 50 ++ .../subagent/subagent-inprocess/src/index.ts | 22 +- scripts/type-equiv.manifest.json | 449 +++++++++++++++--- 5 files changed, 495 insertions(+), 95 deletions(-) diff --git a/docs/core-data-structures/tools.md b/docs/core-data-structures/tools.md index a05ffb3966..a38a5b9c15 100644 --- a/docs/core-data-structures/tools.md +++ b/docs/core-data-structures/tools.md @@ -138,6 +138,40 @@ type PostToolDecision = Call `next()` to delegate to the default (allow / accept-unchanged), or return a decision to short-circuit. A `pre-execute` `deny` (or `ask`, which degrades to deny until the permission system lands) skips dispatch and yields an `isError` result; input rewrite is deliberately NOT offered on `PreToolDecision` (it would desync the pre-execution audit/history/UI from what ran — its own proposed RFC). A `post-execute` `accept` may replace the model-facing `content` (clean, because `tool/result` is logged after `execute()` returns); a `block` turns the call into an `isError` whose content is the corrective `feedback`. Core dispatch sits between the waterfalls as plain code; the tool body keeps its own try/catch so a thrown tool still reaches `post-execute` as an `isError`. An unregistered tool routes through the same catch as a tool-thrown error, so both failure classes get a structured `{ name, code }` (`ToolNotFoundError` → `UNKNOWN_TOOL`) — the loop records a failed tool call instead of failing the whole turn. +## The structured-output schema subset + +The vocabulary a caller uses to demand a machine-readable result from a subagent (`SubagentStartRequest.outputSchema`, [subagent.md](subagent.md#the-start-request)) or a workflow `agent()` call. It is deliberately NOT full JSON Schema: the schema travels verbatim to the model as a forced tool's `parameters`, and the produced value is validated client-side by `validateStructuredValue` — so every accepted keyword must be one the validator actually enforces, and `assertSupportedOutputSchema` rejects anything else loud (`OutputSchemaError`, listing every violation). Both walkers reason over own enumerable properties only (JSON carries nothing else) and reject non-plain objects (`Date`, `Map`) that would serialize lossily. + +```ts type-equiv +type StructuredScalar = string | number | boolean | null +``` + +```ts type-equiv +type StructuredSchemaType = 'object' | 'array' | 'string' | 'number' | 'integer' | 'boolean' | 'null' +``` + +```ts type-equiv +interface StructuredSchemaNode { + type: StructuredSchemaType + properties?: Record + required?: string[] + additionalProperties?: boolean + items?: StructuredSchemaNode + enum?: StructuredScalar[] + const?: StructuredScalar + description?: string + title?: string + default?: unknown + examples?: unknown +} +``` + +A schema is an object-rooted node (`enum`/`const` are scalar-only; `description`/`title`/`default`/`examples` are annotations, allowed and ignored but still required to be JSON data — they ride the wire): + +```ts type-equiv +type StructuredOutputSchema = StructuredSchemaNode & { type: 'object' } +``` + ## Tool-presentation UI vocabulary How a tool wants its call shown in a UI (an editor tool-call card, a CLI log line), provider-neutral so a tool describes itself without depending on any client protocol. `presentCall`/`presentResult` return a **`card`-tagged render intent** — a discriminated union a UI bridge switches on: diff --git a/packages/core/tools/src/json-schema.ts b/packages/core/tools/src/json-schema.ts index 35eeb240a4..bc0da537e1 100644 --- a/packages/core/tools/src/json-schema.ts +++ b/packages/core/tools/src/json-schema.ts @@ -91,9 +91,20 @@ const ANNOTATION_KEYWORDS = new Set(['description', 'title', 'default', 'example const SCHEMA_TYPES: readonly StructuredSchemaType[] = ['object', 'array', 'string', 'number', 'integer', 'boolean', 'null'] -/** Whether a value is a non-null, non-array object (structural, realm-agnostic). */ +/** + * Whether a value is a PLAIN JSON object — non-null, non-array, and with a + * prototype chain of at most one link (`null`-proto, or any realm's + * `Object.prototype`, whose own prototype is `null`). Realm-agnostic on + * purpose: a schema materialized in another realm carries THAT realm's + * `Object.prototype`, which an identity check would wrongly reject. Exotic + * hosts (`Date`, `Map`, class instances) have longer chains and are rejected — + * they would serialize lossily (`Date` → string, `Map` → `{}`) instead of + * failing loud. + */ function isObjectLike(value: unknown): value is Record { - return typeof value === 'object' && value !== null && !Array.isArray(value) + if (typeof value !== 'object' || value === null || Array.isArray(value)) return false + const proto: unknown = Object.getPrototypeOf(value) + return proto === null || Object.getPrototypeOf(proto) === null } /** Whether a value is a supported scalar (`enum`/`const` member): string, finite number, boolean, or null. */ @@ -116,6 +127,9 @@ function isJsonData(value: unknown, seen: Set): boolean { seen.add(value) try { if (Array.isArray(value)) return value.every(entry => isJsonData(entry, seen)) + // A non-plain object (Date, Map, class instance) is NOT JSON data even when + // it has no enumerable values — it would serialize lossily, not loudly. + if (!isObjectLike(value)) return false return Object.values(value).every(entry => isJsonData(entry, seen)) } finally { seen.delete(value) @@ -194,8 +208,11 @@ function checkSchemaNode(node: unknown, path: string, violations: string[], seen violations.push(`${path}.required must be an array of strings`) } else { const declared = isObjectLike(properties) ? properties : {} - for (const key of required) { - if (!(key in declared)) violations.push(`${path}.required names "${key}" which is not in properties`) + // The guard above proved every entry is a string. + for (const key of required as string[]) { + // Own-property check: `in` would let inherited names (`toString`) + // satisfy the declared-in-properties contract via the prototype. + if (!Object.hasOwn(declared, key)) violations.push(`${path}.required names "${key}" which is not in properties`) } } } @@ -256,16 +273,20 @@ function checkValue(node: StructuredSchemaNode, value: unknown, path: string): s if (!isObjectLike(value)) return [`"${path}" must be an object`] const violations: string[] = [] const properties = node.properties ?? {} + // Own-property discipline throughout: JSON carries own enumerable + // properties only, so an inherited `toString` must not satisfy + // `required`, dodge `additionalProperties: false`, or be validated as if + // the value carried it. for (const key of node.required ?? []) { - if (value[key] === undefined) violations.push(`missing required property "${path}.${key}"`) + if (!Object.hasOwn(value, key) || value[key] === undefined) violations.push(`missing required property "${path}.${key}"`) } for (const [key, child] of Object.entries(properties)) { - if (value[key] === undefined) continue + if (!Object.hasOwn(value, key) || value[key] === undefined) continue violations.push(...checkValue(child, value[key], `${path}.${key}`)) } if (node.additionalProperties === false) { for (const key of Object.keys(value)) { - if (!(key in properties)) violations.push(`"${path}.${key}" is not a declared property (additionalProperties: false)`) + if (!Object.hasOwn(properties, key)) violations.push(`"${path}.${key}" is not a declared property (additionalProperties: false)`) } } return violations diff --git a/packages/core/tools/tests/json-schema.spec.ts b/packages/core/tools/tests/json-schema.spec.ts index e7635b06f3..6fa895288e 100644 --- a/packages/core/tools/tests/json-schema.spec.ts +++ b/packages/core/tools/tests/json-schema.spec.ts @@ -154,6 +154,30 @@ describe('assertSupportedOutputSchema', () => { const leaf = { type: 'string' } asserted({ type: 'object', properties: { a: leaf, b: leaf } }) }) + + it('required cannot be satisfied by INHERITED names — `toString` is not a declared property', () => { + // `'toString' in {}` is true via Object.prototype; the declared-property + // contract must be an own-property check. + expect(violationsOf({ type: 'object', properties: {}, required: ['toString'] })) + .toEqual(['schema.required names "toString" which is not in properties']) + }) + + it('rejects exotic host objects where the subset expects plain JSON structure', () => { + // A Map as `properties` has no own enumerable entries: structurally it + // would read as "no properties" and serialize to {} — lossy, not loud. + expect(violationsOf({ type: 'object', properties: new Map() })) + .toEqual(['schema.properties must be an object of schemas']) + // A Date node is not a schema object even though Object.values(date) is []. + expect(violationsOf({ type: 'object', properties: { at: new Date(0) } })) + .toEqual(['schema.properties.at must be a schema object']) + }) + + it('rejects exotic annotation payloads that would serialize lossily', () => { + expect(violationsOf({ type: 'object', default: new Date(0) })) + .toEqual(['schema.default annotation must be JSON data']) + expect(violationsOf({ type: 'object', examples: [new Map()] })) + .toEqual(['schema.examples annotation must be JSON data']) + }) }) describe('validateStructuredValue', () => { @@ -223,6 +247,32 @@ describe('validateStructuredValue', () => { expect(validateStructuredValue(schema, { file: undefined })).toEqual(['missing required property "value.file"']) }) + it('inherited properties satisfy nothing: required, additionalProperties, and recursion are own-property only', () => { + // required: ['toString'] must NOT be satisfied by Object.prototype.toString. + expect(validateStructuredValue( + asserted({ type: 'object', properties: { toString: { type: 'string' } }, required: ['toString'] }), + {}, + )).toEqual(['missing required property "value.toString"']) + // additionalProperties: false must flag an OWN `toString` key even though + // `'toString' in properties` is true via the prototype. + expect(validateStructuredValue( + asserted({ type: 'object', additionalProperties: false }), + { toString: 1 }, + )).toEqual(['"value.toString" is not a declared property (additionalProperties: false)']) + // A declared property the value does NOT carry must not be validated + // against the value's INHERITED member (constructor is a function on + // every plain object's prototype, not a carried property). + expect(validateStructuredValue( + asserted({ type: 'object', properties: { constructor: { type: 'string' } } }), + {}, + )).toEqual([]) + }) + + it('a non-plain object value is not an object in the JSON sense', () => { + expect(validateStructuredValue(asserted({ type: 'object' }), new Date(0))) + .toEqual(['"value" must be an object']) + }) + it('collects multiple violations across branches in one pass', () => { expect(validateStructuredValue(schema, { line: 'x', severity: 'mid' })).toEqual([ 'missing required property "value.file"', diff --git a/packages/subagent/subagent-inprocess/src/index.ts b/packages/subagent/subagent-inprocess/src/index.ts index 72906a78fc..08d240d248 100644 --- a/packages/subagent/subagent-inprocess/src/index.ts +++ b/packages/subagent/subagent-inprocess/src/index.ts @@ -25,11 +25,12 @@ import { type StructuredAcquisition, } from './structured.ts' +// The runtime itself (acquire/attach/release) is package-internal: runs +// acquire it inside startInProcessRun, and no other package drives it. Only +// the model-facing vocabulary is public. export { - acquireStructuredRuntime, STRUCTURED_OUTPUT_TOOL, STRUCTURED_OUTPUT_INSTRUCTION, - type StructuredAcquisition, } from './structured.ts' declare module '@deepseek-ai/dsh-agent' { @@ -110,15 +111,18 @@ export function startInProcessRun( if (request.maxDepth !== undefined && childDepth > request.maxDepth) { throw new SubagentDepthError(childDepth, request.maxDepth) } - // Snapshot, then assert, the schema subset BEFORE any child exists (the + // Assert, then snapshot, the schema subset BEFORE any child exists (the // service has already capability-gated; this rejects a schema outside the - // enforced subset loud). The snapshot is load-bearing: the caller keeps its - // reference, so validating and attaching the ORIGINAL would let a - // post-start() mutation drift the enforced schema away from the asserted - // one — the clone pins assertion, the model-visible parameters, and - // validateStructuredValue to the same isolation-immutable value. + // enforced subset loud). Assertion comes FIRST so a hostile value fails as + // OutputSchemaError, never as structuredClone's raw DataCloneError — the + // asserted subset is plain JSON data, which always clones. The snapshot is + // load-bearing: the caller keeps its reference, so attaching the ORIGINAL + // would let a post-start() mutation drift the enforced schema away from the + // asserted one — the clone (taken synchronously with the assertion, no + // interleaving possible) pins assertion, the model-visible parameters, and + // validateStructuredValue to one isolation-immutable value. + if (request.outputSchema !== undefined) assertSupportedOutputSchema(request.outputSchema) const schema = request.outputSchema === undefined ? undefined : structuredClone(request.outputSchema) - if (schema !== undefined) assertSupportedOutputSchema(schema) const childId = AgentId(randomUUID()) // The child's OWN events begin after the seed (fork seeds the parent's diff --git a/scripts/type-equiv.manifest.json b/scripts/type-equiv.manifest.json index b5c648527e..e414cff1f4 100644 --- a/scripts/type-equiv.manifest.json +++ b/scripts/type-equiv.manifest.json @@ -1,84 +1,375 @@ { "comment": "Maps each ` ```ts type-equiv ` block (by doc + declared symbol) to the source symbol it must match verbatim. verify-type-equiv.ts enforces a 1:1 correspondence: every type-equiv block has exactly one entry here, and every entry resolves to exactly one block. Add an entry when you add a type-equiv block; remove it when you remove the block.", "entries": [ - { "doc": "docs/core-data-structures/core.md", "symbol": "Branded", "source": "packages/util/brand/src/index.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "ContentBlockMap", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "Message", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "MessageSourceMap", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "FinishReasonMap", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "GenerateOptions", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "ToolSchema", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "LlmCallConfig", "source": "packages/llm/llm/src/call-config.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "Agent", "source": "packages/core/agent/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "HookContext", "source": "packages/core/agent/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "PromptDecision", "source": "packages/core/agent/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "ContinuationDecision", "source": "packages/core/agent/src/types.ts" }, - { "doc": "docs/core-data-structures/core.md", "symbol": "SessionStartSource", "source": "packages/core/agent/src/types.ts" }, - - { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "StreamChunk", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "TokenUsage", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "ContentBlockMap", "source": "packages/llm/llm/src/types.ts" }, - { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "AppIdentity", "source": "packages/llm/llm/src/attribution.ts" }, - - { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEventMap", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "EpochHeader", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "TodoItem", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "TurnTriggerMap", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "TurnEndReasonMap", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceEventType", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceOp", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceIntent", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceNode", "source": "packages/core/session/src/surface.ts" }, - - { "doc": "docs/core-data-structures/persistence.md", "symbol": "SessionHeader", "source": "packages/core/session/src/types.ts" }, - { "doc": "docs/core-data-structures/persistence.md", "symbol": "CreateSessionOptions", "source": "packages/core/session/src/types.ts" }, - - { "doc": "docs/core-data-structures/tools.md", "symbol": "ToolDefinition", "source": "packages/core/tools/src/index.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "SchemaProp", "source": "packages/core/tools/src/schema.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "SchemaSpec", "source": "packages/core/tools/src/schema.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "InferArgs", "source": "packages/core/tools/src/schema.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "ToolExecution", "source": "packages/core/tools/src/index.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "ToolExecutionResult", "source": "packages/core/tools/src/index.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "PreToolDecision", "source": "packages/core/tools/src/index.ts" }, - { "doc": "docs/core-data-structures/tools.md", "symbol": "PostToolDecision", "source": "packages/core/tools/src/index.ts" }, - - { "doc": "docs/core-data-structures/bash.md", "symbol": "BashExecRequest", "source": "packages/bash/bash/src/types.ts" }, - { "doc": "docs/core-data-structures/bash.md", "symbol": "BashExecSpec", "source": "packages/bash/bash/src/types.ts" }, - { "doc": "docs/core-data-structures/bash.md", "symbol": "BashRunResult", "source": "packages/bash/bash/src/types.ts" }, - { "doc": "docs/core-data-structures/bash.md", "symbol": "CollectedOutput", "source": "packages/bash/bash/src/types.ts" }, - { "doc": "docs/core-data-structures/bash.md", "symbol": "BashTask", "source": "packages/bash/bash/src/types.ts" }, - { "doc": "docs/core-data-structures/bash.md", "symbol": "BashTaskRead", "source": "packages/bash/bash/src/types.ts" }, - - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsTarget", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsTargetKey", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsVersion", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsInfo", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsDirEntry", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsWriteIntent", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsWriteOutcome", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsEditRequest", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsEditOutcome", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsErrorCode", "source": "packages/fs/fs/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsPolicyExec", "source": "packages/fs/fs-policy/src/types.ts" }, - { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FileReadOutcome", "source": "packages/fs/tool-fs/src/read-render.ts" }, - - { "doc": "docs/core-data-structures/compaction.md", "symbol": "CompactionResult", "source": "packages/compact/compact/src/types.ts" }, - - { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentCapabilities", "source": "packages/subagent/subagent/src/types.ts" }, - { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentStartRequest", "source": "packages/subagent/subagent/src/types.ts" }, - { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentResult", "source": "packages/subagent/subagent/src/types.ts" }, - { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentStopReasonMap", "source": "packages/subagent/subagent/src/types.ts" }, - { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentRun", "source": "packages/subagent/subagent/src/types.ts" }, - { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentProvider", "source": "packages/subagent/subagent/src/types.ts" }, - - { "doc": "docs/core-data-structures/web.md", "symbol": "WebSearchRequest", "source": "packages/web/web/src/types.ts" }, - { "doc": "docs/core-data-structures/web.md", "symbol": "WebSearchResult", "source": "packages/web/web/src/types.ts" }, - { "doc": "docs/core-data-structures/web.md", "symbol": "WebSearchSource", "source": "packages/web/web/src/types.ts" }, - { "doc": "docs/core-data-structures/web.md", "symbol": "WebFetchRequest", "source": "packages/web/web/src/types.ts" }, - { "doc": "docs/core-data-structures/web.md", "symbol": "WebFetchResult", "source": "packages/web/web/src/types.ts" }, - { "doc": "docs/core-data-structures/web.md", "symbol": "WebFetchBody", "source": "packages/web/web/src/types.ts" }, - { "doc": "docs/core-data-structures/web.md", "symbol": "WebProviderStatus", "source": "packages/web/web/src/types.ts" } + { + "doc": "docs/core-data-structures/core.md", + "symbol": "Branded", + "source": "packages/util/brand/src/index.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "ContentBlockMap", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "Message", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "MessageSourceMap", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "FinishReasonMap", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "GenerateOptions", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "ToolSchema", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "LlmCallConfig", + "source": "packages/llm/llm/src/call-config.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "SessionEvent", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "Agent", + "source": "packages/core/agent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "HookContext", + "source": "packages/core/agent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "PromptDecision", + "source": "packages/core/agent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "ContinuationDecision", + "source": "packages/core/agent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/core.md", + "symbol": "SessionStartSource", + "source": "packages/core/agent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/llm-streaming.md", + "symbol": "StreamChunk", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/llm-streaming.md", + "symbol": "TokenUsage", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/llm-streaming.md", + "symbol": "ContentBlockMap", + "source": "packages/llm/llm/src/types.ts" + }, + { + "doc": "docs/core-data-structures/llm-streaming.md", + "symbol": "AppIdentity", + "source": "packages/llm/llm/src/attribution.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "SessionEventMap", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "EpochHeader", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "TodoItem", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "SessionEvent", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "TurnTriggerMap", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "TurnEndReasonMap", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "SurfaceEventType", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "SurfaceOp", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "SurfaceIntent", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/session.md", + "symbol": "SurfaceNode", + "source": "packages/core/session/src/surface.ts" + }, + { + "doc": "docs/core-data-structures/persistence.md", + "symbol": "SessionHeader", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/persistence.md", + "symbol": "CreateSessionOptions", + "source": "packages/core/session/src/types.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "ToolDefinition", + "source": "packages/core/tools/src/index.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "SchemaProp", + "source": "packages/core/tools/src/schema.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "SchemaSpec", + "source": "packages/core/tools/src/schema.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "InferArgs", + "source": "packages/core/tools/src/schema.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "ToolExecution", + "source": "packages/core/tools/src/index.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "ToolExecutionResult", + "source": "packages/core/tools/src/index.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "PreToolDecision", + "source": "packages/core/tools/src/index.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "PostToolDecision", + "source": "packages/core/tools/src/index.ts" + }, + { + "doc": "docs/core-data-structures/bash.md", + "symbol": "BashExecRequest", + "source": "packages/bash/bash/src/types.ts" + }, + { + "doc": "docs/core-data-structures/bash.md", + "symbol": "BashExecSpec", + "source": "packages/bash/bash/src/types.ts" + }, + { + "doc": "docs/core-data-structures/bash.md", + "symbol": "BashRunResult", + "source": "packages/bash/bash/src/types.ts" + }, + { + "doc": "docs/core-data-structures/bash.md", + "symbol": "CollectedOutput", + "source": "packages/bash/bash/src/types.ts" + }, + { + "doc": "docs/core-data-structures/bash.md", + "symbol": "BashTask", + "source": "packages/bash/bash/src/types.ts" + }, + { + "doc": "docs/core-data-structures/bash.md", + "symbol": "BashTaskRead", + "source": "packages/bash/bash/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsTarget", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsTargetKey", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsVersion", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsInfo", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsDirEntry", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsWriteIntent", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsWriteOutcome", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsEditRequest", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsEditOutcome", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsErrorCode", + "source": "packages/fs/fs/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FsPolicyExec", + "source": "packages/fs/fs-policy/src/types.ts" + }, + { + "doc": "docs/core-data-structures/filesystem.md", + "symbol": "FileReadOutcome", + "source": "packages/fs/tool-fs/src/read-render.ts" + }, + { + "doc": "docs/core-data-structures/compaction.md", + "symbol": "CompactionResult", + "source": "packages/compact/compact/src/types.ts" + }, + { + "doc": "docs/core-data-structures/subagent.md", + "symbol": "SubagentCapabilities", + "source": "packages/subagent/subagent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/subagent.md", + "symbol": "SubagentStartRequest", + "source": "packages/subagent/subagent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/subagent.md", + "symbol": "SubagentResult", + "source": "packages/subagent/subagent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/subagent.md", + "symbol": "SubagentStopReasonMap", + "source": "packages/subagent/subagent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/subagent.md", + "symbol": "SubagentRun", + "source": "packages/subagent/subagent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/subagent.md", + "symbol": "SubagentProvider", + "source": "packages/subagent/subagent/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebSearchRequest", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebSearchResult", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebSearchSource", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebFetchRequest", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebFetchResult", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebFetchBody", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/web.md", + "symbol": "WebProviderStatus", + "source": "packages/web/web/src/types.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "StructuredScalar", + "source": "packages/core/tools/src/json-schema.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "StructuredSchemaType", + "source": "packages/core/tools/src/json-schema.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "StructuredSchemaNode", + "source": "packages/core/tools/src/json-schema.ts" + }, + { + "doc": "docs/core-data-structures/tools.md", + "symbol": "StructuredOutputSchema", + "source": "packages/core/tools/src/json-schema.ts" + } ] } From 280233ba781034fcdd662377385676609e3c2cc9 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 21:08:12 +0800 Subject: [PATCH 14/24] fix review finding: the capture commits only on the final post-execute accept MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The cross-seam blocker: structured_output recorded its value in the tool BODY, before tools/post-execute could block the call — a PostToolUse hook's block turned the logged result into isError while readResult still returned structured success and the continuation veto ended the turn. Two-phase commit: the body validates and STAGES (RunState.pending); a fourth runtime listener on tools/post-execute — prepend, so await next() returns the composed final decision — promotes the stage to captured only on an accepted call, and clears it on every path. A block now yields a consistent pair: the model and log see the isError feedback, the run settles error with no structured value, and the turn continues so the model can react. Regressions: block denies the capture end-to-end; accept-with-replacement still commits. --- .../subagent-inprocess/src/structured.ts | 64 +++++- .../tests/structured.spec.ts | 186 ++++++++++++------ 2 files changed, 179 insertions(+), 71 deletions(-) diff --git a/packages/subagent/subagent-inprocess/src/structured.ts b/packages/subagent/subagent-inprocess/src/structured.ts index 370f7a538d..a557e44785 100644 --- a/packages/subagent/subagent-inprocess/src/structured.ts +++ b/packages/subagent/subagent-inprocess/src/structured.ts @@ -36,14 +36,20 @@ * closes the within-step window the continuation veto cannot: a * `tools/pre-execute` deny for any call arriving after the agent's capture, so * a response that lists `structured_output` before further tool calls cannot - * run side effects after the final answer was accepted. + * run side effects after the final answer was accepted. A fourth, + * `tools/post-execute`, is the capture COMMIT: the tool body only stages the + * validated value, and it becomes the run's captured result only when the + * final post-execute decision accepts the call — a blocking hook downstream + * yields `isError` in the log, and the run must not report success for it. * - * Lifetime is refcounted with two kinds of holder: each backend acquires for - * its plugin lifetime (so the tool exists before any run), and each structured - * RUN acquires from start to settle (so a backend hot-reload mid-run cannot - * unregister the capture tool out from under a live child). Registrations are - * effects on the ROOT context — their natural upper bound is app teardown — and - * the refcount disposes them when the last holder releases. + * Lifetime is refcounted by structured RUNS: each acquires from start to + * settle, so the registrations exist exactly while at least one structured + * child is live — a plain deployment that never passes `outputSchema` carries + * no always-on global state, and a backend hot-reload mid-run cannot + * unregister the capture tool out from under a live child (the run holds its + * own acquisition). Registrations land on the ROOT context and the refcount + * disposes them when the last run settles; the next structured run + * re-registers them. * * @module @deepseek-ai/dsh-subagent-inprocess/structured */ @@ -53,7 +59,7 @@ import type { Agent } from '@deepseek-ai/dsh-agent' import type { ContentBlock, ToolSchema } from '@deepseek-ai/dsh-llm' import type { ContinuationDecision } from '@deepseek-ai/dsh-agent' import type { AssembleContext, PromptAssembly } from '@deepseek-ai/dsh-system-prompt' -import type { PreToolDecision, ToolExecution } from '@deepseek-ai/dsh-tools' +import type { PostToolDecision, PreToolDecision, ToolExecution, ToolExecutionResult } from '@deepseek-ai/dsh-tools' import { ToolArgsError, validateStructuredValue, type StructuredOutputSchema } from '@deepseek-ai/dsh-tools' /** The model-facing tool name a structured child must call to finish. */ @@ -75,6 +81,15 @@ export const STRUCTURED_OUTPUT_INSTRUCTION /** One structured run's state: the schema to enforce and the captured value, once recorded. */ interface RunState { readonly schema: StructuredOutputSchema + /** + * A validated value awaiting the post-execute verdict on ITS OWN call. Set + * by the capture tool's body, promoted to {@link RunState.captured} only + * when the final `tools/post-execute` decision accepts the call — a + * downstream block turns the logged result into `isError`, and a value + * committed at body time would let the run report success for a call the + * model saw fail. + */ + pending?: { value: unknown } captured?: { value: unknown } } @@ -106,7 +121,7 @@ export interface StructuredAcquisition { /** * Acquire the per-root-context structured runtime, registering the capture tool - * and the two waterfall listeners on the FIRST acquisition. See the module doc + * and the runtime's listeners on the FIRST acquisition. See the module doc * for the enforcement and lifetime design. * @param ctx - any context of the app; the runtime keys off `ctx.root`. * @returns this holder's handle (attach/captured/detach + idempotent release). @@ -179,7 +194,9 @@ function registerRuntime(root: Context, runtime: StructuredRuntime): void { // ToolArgsError → isError result with INVALID_ARGS: the model retries // within the same turn, exactly like a schema-validated defineTool call. if (violations.length > 0) throw new ToolArgsError(violations) - state.captured = { value: args } + // Two-phase commit: the body only STAGES the value; the post-execute + // listener below promotes it once the final decision accepts the call. + state.pending = { value: args } return Promise.resolve([{ type: 'text', text: 'Structured output recorded.' }]) }, }) @@ -243,6 +260,33 @@ function registerRuntime(root: Context, runtime: StructuredRuntime): void { return next() }, { prepend: true })) + // The capture COMMIT: promote the staged value only when the final + // post-execute decision accepts the call. The capture tool's body cannot + // decide — `tools/post-execute` runs after it, and a blocking listener (a + // PostToolUse hook) turns the logged result into `isError` feedback; a value + // committed at body time would make readResult report `structured` success + // for a call whose result the model and session log saw fail. `prepend: + // true` = outermost at registration time, so `await next()` returns the + // COMPOSED downstream decision — the same final verdict the registry maps + // onto the result. (A later-registered outer listener that blocks without + // delegating skips this commit entirely: the staged value is dropped and the + // run errors — failure-safe in the same direction.) The staging slot clears + // on every path, including a rejecting downstream listener. + runtime.disposers.push(root.on('tools/post-execute', async function ( + this: unknown, exec: ToolExecution, _result: ToolExecutionResult, next: () => Promise, + ): Promise { + const state = exec.agent ? runtime.states.get(exec.agent) : undefined + if (!state || exec.name !== STRUCTURED_OUTPUT_TOOL || state.pending === undefined) return next() + const pending = state.pending + try { + const decision = await next() + if (decision.kind === 'accept') state.captured = pending + return decision + } finally { + delete state.pending + } + }, { prepend: true })) + // Terminal means terminal WITHIN the step, not only at its end: the // turn-continuation veto above runs after every call in the current model // response has executed, so a response that puts `structured_output` before diff --git a/packages/subagent/subagent-inprocess/tests/structured.spec.ts b/packages/subagent/subagent-inprocess/tests/structured.spec.ts index ba693cfecd..0de036a0a0 100644 --- a/packages/subagent/subagent-inprocess/tests/structured.spec.ts +++ b/packages/subagent/subagent-inprocess/tests/structured.spec.ts @@ -11,8 +11,7 @@ import * as Invariants from '@deepseek-ai/dsh-invariants' import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent' import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools' import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts' -import * as spawn from '@deepseek-ai/dsh-subagent-spawn' -import * as fork from '@deepseek-ai/dsh-subagent-fork' +import { startInProcessRun } from '../src/index.ts' import { acquireStructuredRuntime, STRUCTURED_OUTPUT_INSTRUCTION, @@ -28,11 +27,14 @@ const SCHEMA: StructuredOutputSchema = { } /** - * Real loop + scripted mock model + the REAL spawn backend (which acquires the - * structured runtime at apply, exactly as shipped). The mock model script - * drives the child's structured_output calls. + * Real loop + scripted mock model + an INLINE spawn-shaped provider over the + * shared driver. The concrete backend plugins are deliberately NOT loaded — + * they would devDep-cycle this package (spawn/fork already depend on the + * driver), and the runtime under test is the driver's; plugin-level structured + * coverage lives in the spawn/fork specs. The mock model script drives the + * child's structured_output calls. */ -async function setup(script: Script, options?: { withFork?: boolean }) { +async function setup(script: Script) { const ctx = new Context() const adapter = new MockAdapter(script) await ctx.plugin(LlmService) @@ -43,13 +45,15 @@ async function setup(script: Script, options?: { withFork?: boolean }) { await ctx.plugin(Invariants) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SubagentService) - const fiber = await ctx.plugin(spawn, { providerName: 'spawn' }) - const forkFiber = options?.withFork - ? await ctx.plugin(fork, { providerName: 'fork' }) - : undefined + const disposeProvider = ctx.subagents.registerProvider({ + name: 'spawn', + capabilities: { outputSchema: true, depthLimit: true, toolFilter: false }, + inheritsParentContext: false, + start: (request: SubagentStartRequest) => startInProcessRun(ctx, request, { providerName: 'spawn' }), + }) ctx.llm.registerAdapter(['mock'], adapter) const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) - return { ctx, parent, adapter, fiber, forkFiber } + return { ctx, parent, adapter, disposeProvider } } function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial): SubagentStartRequest { @@ -263,6 +267,62 @@ describe('in-process structured output', () => { expect(ctx.agents.get(AgentId('parent'))).toBeDefined() }) + it('a schema carrying non-JSON values fails as OutputSchemaError, never as a raw clone error', async () => { + const { ctx, parent } = await setup([]) + // Assertion runs BEFORE the defensive structuredClone: a function-valued + // annotation must surface as the subset violation it is, not escape as + // structuredClone's DataCloneError. + expect(() => ctx.subagents.start('spawn', structuredRequest(parent, { + outputSchema: { type: 'object', default: () => {} } as unknown as StructuredOutputSchema, + }))).toThrow(/unsupported output schema.*annotation must be JSON data/) + }) + + it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => { + const { ctx, parent, adapter } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }), + textResponse('continues after the blocked capture'), + ]) + // A PostToolUse-style hook, registered AFTER the runtime (so the runtime's + // prepend commit listener stays outermost and composes this verdict). + ctx.on('tools/post-execute', (exec, _result, next) => { + if (exec.name === STRUCTURED_OUTPUT_TOOL) { + return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] }) + } + return next() + }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + // No capture was committed: the run reports the schema shortfall... + expect(result.structured).toBeUndefined() + expect(result.stopReason).toBe('error') + // ...the logged tool result is the blocked isError with the feedback... + const child = ctx.agents.get(run.id)! + const results = child.session.events.filter(e => e.type === 'tool/result') + expect((results[0]!.data as { isError?: boolean }).isError).toBe(true) + expect(JSON.stringify((results[0]!.data as { content: unknown }).content)).toContain('capture rejected by hook') + // ...and the turn CONTINUED past the blocked call (no captured veto): + // the model got to react to the failure with a second step. + expect(adapter.requests.length).toBe(2) + await run.dispose() + }) + + it('a post-execute accept-with-replacement still commits the capture', async () => { + const { ctx, parent } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }), + ]) + ctx.on('tools/post-execute', (exec, _result, next) => { + if (exec.name === STRUCTURED_OUTPUT_TOOL) { + return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] }) + } + return next() + }) + const run = ctx.subagents.start('spawn', structuredRequest(parent)) + const result = await run.result + expect(result.stopReason).toBe('completed') + expect(result.structured).toEqual({ answer: 8 }) + await run.dispose() + }) + it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => { const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })]) // A context-wide section stands in for the deployment persona: the @@ -298,6 +358,21 @@ describe('in-process structured output', () => { }) describe('final-request enforcement (the prepend agent/request listener)', () => { + it('a plain agent assembling while the runtime is LIVE gets the placeholder stripped', async () => { + // Run-scoped acquisition means a plain deployment never registers the + // tool at all; the strip branch exists for the CONCURRENT case — a plain + // agent taking a turn while some structured child holds the runtime open. + const { ctx, parent, adapter } = await setup([textResponse('parent answer')]) + const hold = acquireStructuredRuntime(ctx) + parent.send([{ type: 'text', text: 'hello' }]) + await parent.whenIdle() + // The placeholder IS in the registry during this turn; the assembly the + // loop rendered must not carry it for an agent without a structured run. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL) + hold.release() + }) + it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => { const { ctx, parent, adapter } = await setup([ // Parent turn (a plain agent): must NOT see the tool. @@ -396,10 +471,13 @@ describe('in-process structured output', () => { // and shape a structured agent's assembly on the same path the loop // renders and logs as the request header. const { ctx, parent } = await setup([]) + const acquisition = acquireStructuredRuntime(ctx) + // Bare assemble WHILE the runtime is live: the no-agent branch must + // strip the registered placeholder (before the acquisition there is + // nothing to strip — run-scoped registration). const bare = await ctx.systemPrompt.assemble({}) expect(bare.tools.map(tool => tool.name)).not.toContain(STRUCTURED_OUTPUT_TOOL) - const acquisition = acquireStructuredRuntime(ctx) acquisition.attach(parent, SCHEMA) const shaped = await ctx.systemPrompt.assemble({ agent: parent }) expect(shaped.tools.map(tool => tool.name)).toContain(STRUCTURED_OUTPUT_TOOL) @@ -412,57 +490,37 @@ describe('in-process structured output', () => { }) }) - describe('runtime lifetime (refcount: backends + live runs)', () => { - it('registers the capture tool while a backend is loaded and unregisters when the last unloads', async () => { - const { ctx, fiber, forkFiber } = await setup([], { withFork: true }) - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() - await fiber.dispose() - // fork still holds a reference. - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() - await forkFiber!.dispose() - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() - }) - - it('a live run-level acquisition keeps the runtime registered after EVERY backend unloads', async () => { - // Simulates the run-holder half of the two-level lifetime: a structured - // run acquires at start and releases at settle, so registration ordering - // is settle-then-unregister even if all backends unload first. (A real - // in-process child dies WITH its backend's fiber — the acquisition's - // observable job is this ordering, which a manual holder pins directly.) - const { ctx, fiber, forkFiber } = await setup([], { withFork: true }) - const runHolder = acquireStructuredRuntime(ctx) - await fiber.dispose() - await forkFiber!.dispose() - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() - runHolder.release() - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() - }) - - it('a structured run releases its acquisition when it settles (backend unload mid-run)', async () => { - const { ctx, parent, fiber } = await setup(['hang']) - const run = ctx.subagents.start('spawn', structuredRequest(parent)) - // Let the child's step start streaming, then unload the backend. The - // backend owns the child agent, so the unload tears the child down and - // the run settles — releasing its own acquisition on the way out. - await new Promise(resolve => setTimeout(resolve, 30)) - await fiber.dispose() - const result = await run.result - expect(result.stopReason).toBe('error') - // Both holders (backend + run) released — nothing keeps the runtime now. - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() - await run.dispose() - }) - - it('fork children capture structured output through the same runtime', async () => { + describe('runtime lifetime (refcount: live structured runs)', () => { + it('the runtime exists exactly while structured runs are live: nothing before, nothing after', async () => { const { ctx, parent } = await setup([ - toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }), - ], { withFork: true }) - const run = ctx.subagents.start('fork', structuredRequest(parent)) + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }), + ]) + // No always-on global state: a context that has run no structured child + // carries no capture tool. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + const run = ctx.subagents.start('spawn', structuredRequest(parent)) const result = await run.result - expect(result.structured).toEqual({ answer: 9 }) + // The capture succeeded — the registrations existed while the run lived. + expect(result.structured).toEqual({ answer: 4 }) + // The run's settle released the last acquisition. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() await run.dispose() }) + it('concurrent structured runs share one runtime; the last settle disposes it', async () => { + const { ctx, parent } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }), + toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 2 }), + ]) + const first = ctx.subagents.start('spawn', structuredRequest(parent)) + const second = ctx.subagents.start('spawn', structuredRequest(parent)) + const [a, b] = await Promise.all([first.result, second.result]) + expect([a.structured, b.structured].sort()).toEqual([{ answer: 1 }, { answer: 2 }].sort()) + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + await first.dispose() + await second.dispose() + }) + it('acquisition release is idempotent (double release cannot underflow the refcount)', async () => { const ctx = new Context() await ctx.plugin(SystemPrompt) @@ -513,13 +571,16 @@ describe('in-process structured output', () => { acquisition.detach(parent) acquisition.detach(parent) acquisition.release() - // The backend still holds its own reference from setup(). - expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined() + // That manual acquisition was the ONLY holder - release disposes. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() }) }) it('a direct structured_output call from an agent WITHOUT a structured run is an isError', async () => { const { ctx, parent } = await setup([]) + // Hold the runtime open (run-scoped: nothing is registered otherwise) so + // the call reaches the capture tool's own fail-loud guard, not UNKNOWN_TOOL. + const hold = acquireStructuredRuntime(ctx) const result = await ctx.tools.execute({ callId: 'x' as never, name: STRUCTURED_OUTPUT_TOOL, @@ -527,16 +588,19 @@ describe('in-process structured output', () => { agent: parent, }) expect(result.isError).toBe(true) - expect(result.content[0]).toMatchObject({ type: 'text' }) + expect(JSON.stringify(result.content)).toContain('only available to subagents') + hold.release() }) it('a structured_output call with NO calling agent at all is an isError', async () => { const { ctx } = await setup([]) + const hold = acquireStructuredRuntime(ctx) const result = await ctx.tools.execute({ callId: 'x' as never, name: STRUCTURED_OUTPUT_TOOL, arguments: { answer: 1 }, }) expect(result.isError).toBe(true) + hold.release() }) }) From 0e0f3b2f19f9768b2518fca636f882c13bbc60c8 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 21:09:02 +0800 Subject: [PATCH 15/24] review: acquire the structured runtime per run, not per backend MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Codex simplification concern plus the duplication comment on the spawn apply, resolved by deletion: the backend-lifetime holds are gone, so the runtime registers at the first structured run and disposes when the last settles — a deployment that never passes outputSchema carries no always-on global state, and there is no per-backend acquisition block left to extract. The driver spec now drives an INLINE spawn-shaped provider over startInProcessRun, which removes the spawn/fork devDependencies (the test-only workspace cycle); plugin-level structured coverage moves to the backends' own specs (capture through the shipped plugin, mid-run backend unload, seeded fork capture). tools.md, the driver README, and both backend READMEs describe the run-scoped lifetime; the module-graph regenerates without the cycle edges. --- packages/subagent/subagent-fork/src/index.ts | 15 ++--- .../subagent-fork/tests/subagent-fork.spec.ts | 29 +++++++--- .../subagent/subagent-inprocess/README.md | 14 +++-- .../subagent/subagent-inprocess/package.json | 2 - packages/subagent/subagent-spawn/README.md | 2 +- packages/subagent/subagent-spawn/src/index.ts | 20 ++----- .../tests/subagent-spawn.spec.ts | 58 ++++++++++++++++--- pnpm-lock.yaml | 6 -- 8 files changed, 91 insertions(+), 55 deletions(-) diff --git a/packages/subagent/subagent-fork/src/index.ts b/packages/subagent/subagent-fork/src/index.ts index 4ee28001d2..8f91186cf0 100644 --- a/packages/subagent/subagent-fork/src/index.ts +++ b/packages/subagent/subagent-fork/src/index.ts @@ -25,13 +25,13 @@ import z from 'schemastery' import type { SessionEvent } from '@deepseek-ai/dsh-session' import type { Agent } from '@deepseek-ai/dsh-agent' import type { SubagentCapabilities, SubagentProvider, SubagentStartRequest } from '@deepseek-ai/dsh-subagent' -import { acquireStructuredRuntime, startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' +import { startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' export const name = 'subagent-fork' // `tools` is deliberately NOT injected — same rationale as subagent-spawn: the -// structured runtime gates its capture-tool registration on `tools` itself, so -// this backend's apply timing (and the delegation tool's position in the -// model-visible tool list) is unchanged by structured output. +// per-run structured runtime gates its capture-tool registration on `tools` +// itself, so this backend's apply timing (and the delegation tool's position +// in the model-visible tool list) is unchanged by structured output. export const inject = ['subagents', 'agents'] /** Config: the registry name to register the provider under. */ @@ -84,12 +84,5 @@ class ForkProvider implements SubagentProvider { } export function apply(ctx: Context, config: Config): void { - // Hold the structured runtime for the plugin's lifetime (see the spawn - // backend — same two-level lifetime: backends for availability, runs for - // mid-run survival across a backend unload). - ctx.effect(() => { - const acquisition = acquireStructuredRuntime(ctx) - return () => { acquisition.release() } - }, 'subagent-fork structured runtime') ctx.subagents.registerProvider(new ForkProvider(config.providerName, ctx)) } diff --git a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts index 74974942b5..90e2d583d8 100644 --- a/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts +++ b/packages/subagent/subagent-fork/tests/subagent-fork.spec.ts @@ -9,9 +9,10 @@ import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import * as Invariants from '@deepseek-ai/dsh-invariants' import SubagentService from '@deepseek-ai/dsh-subagent' -import { MockAdapter, textResponse } from '../../../core/agent-loop/tests/mock-adapter.ts' +import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts' import type { StreamChunk } from '@deepseek-ai/dsh-llm' import * as fork from '../src/index.ts' +import { STRUCTURED_OUTPUT_TOOL } from '@deepseek-ai/dsh-subagent-inprocess' import { completedTurnPrefix } from '../src/index.ts' type Script = ConstructorParameters[0] @@ -141,6 +142,26 @@ describe('dsh-subagent-fork', () => { await run.dispose() }) + it('captures structured output through the shipped plugin (seeded child, driver runtime)', async () => { + const { ctx, parent } = await setup([ + textResponse('parent turn'), + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }), + ]) + parent.send([{ type: 'text', text: 'warm up' }]) + await parent.whenIdle() + const run = ctx.subagents.start('fork', { + prompt: [{ type: 'text', text: 'report structured' }], + parent, + outputSchema: { type: 'object', properties: { answer: { type: 'number' } }, required: ['answer'] }, + }) + const result = await run.result + expect(result.stopReason).toBe('completed') + expect(result.structured).toEqual({ answer: 9 }) + // Run-scoped runtime: nothing stays registered after the settle. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + await run.dispose() + }) + it('does NOT return the seeded parent output when the child produces no message of its own', async () => { // Regression: readResult must scope to the child's OWN events (after the // seed). The parent completes a turn with a distinctive assistant message, @@ -170,12 +191,6 @@ describe('dsh-subagent-fork', () => { const ctx = new Context() await ctx.plugin(SubagentService) await ctx.plugin(AgentRegistry) - // The backend does NOT inject 'tools' (the structured runtime gates its - // capture-tool registration on tools availability itself, keeping backend - // apply timing — and the delegation tool's prompt position — unchanged); - // the registries are loaded here so the runtime registers eagerly anyway. - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) const fiber = await ctx.plugin(fork, { providerName: 'fork' }) expect(ctx.subagents.list()).toEqual(['fork']) await fiber.dispose() diff --git a/packages/subagent/subagent-inprocess/README.md b/packages/subagent/subagent-inprocess/README.md index a4fcc51fbc..f6870929a8 100644 --- a/packages/subagent/subagent-inprocess/README.md +++ b/packages/subagent/subagent-inprocess/README.md @@ -8,7 +8,7 @@ The shared **in-process subagent run driver**. A pure library (no provider, no r Runs a child as a child [`Agent`](../../core/agent) on the same cordis context (`ctx.agents`): -1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability); a `request.outputSchema` is asserted against the supported subset (`assertSupportedOutputSchema` from [dsh-tools](../../core/tools/README.md)) before any child exists; +1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability); a `request.outputSchema` is asserted against the supported subset (`assertSupportedOutputSchema` from [dsh-tools](../../core/tools/README.md)) and then snapshotted with `structuredClone` before any child exists — assertion first so a hostile value fails as `OutputSchemaError` (never a raw clone error), the snapshot so a post-`start()` caller mutation cannot drift the enforced schema; 2. creates a child via `ctx.agents.create` with a fresh `AgentId`/`SessionId`, the parent's `cwd` + `parentSession` lineage, the optional `options.seed` (fork's completed-turn prefix; omitted for a fresh child), and `agentOptions` (the child inherits the **parent's model** by default — a child with no model can't run — overridable via `request.agentOptions.model`; the deployment persona needs no inheritance — it is a context-wide prompt section); 3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts); there is deliberately NO re-prompt for a structured child that finished cleanly without calling `structured_output` — the shortfall maps to an `error` result for the parent; 4. reads the result, scoped to the child's OWN events (everything at or after `seedLength`, so a seeded child that produced no message of its own never returns the seeded parent's last message): the last `assistant/message` content (deep-cloned — the log is frozen) and the last `turn/end.reason` mapped to a `SubagentStopReason`. A structured run surfaces the captured value as `result.structured`; a structured child that finished cleanly WITHOUT ever capturing settles `error` (a clean finish without the demanded result is a failure, not a success with a missing field). @@ -19,16 +19,18 @@ Runs a child as a child [`Agent`](../../core/agent) on the same cordis context ( `{ providerName: string; seed?: SessionEvent[] }` — the per-backend inputs: the provider name (for error context) and the optional child-session seed. -### Structured output: `acquireStructuredRuntime(ctx): StructuredAcquisition` +### Structured output (package-internal runtime) -The mechanism behind `outputSchema` for in-process children. One globally registered `structured_output` capture tool (its registered parameters are a placeholder) plus two listeners, registered once per root context and shared by every holder: +The mechanism behind `outputSchema` for in-process children — acquired per structured RUN inside `startInProcessRun` (nothing is registered on a context that never runs a structured child; only the model-facing constants `STRUCTURED_OUTPUT_TOOL`/`STRUCTURED_OUTPUT_INSTRUCTION` are exported). One globally registered `structured_output` capture tool (its registered parameters are a placeholder) plus four listeners: -- an `agent/request` waterfall listener registered `prepend: true` that post-processes `await next()` — **final-request enforcement**: the request that hits the wire never carries `structured_output` for an agent without a structured run, and for one that has it always carries the run's OWN schema (as the tool's `parameters`) plus the calling instruction appended to its `system` text (the demand travels with the tool — `AgentOptions` has no per-agent prompt field to carry it). Per-agent shaping lives here because the tool registry and prompt assembly are context-global while schemas differ per concurrent child; cooperative mutate-then-`next()` would not survive a downstream listener returning a replacement request. +- a `system-prompt/assemble` waterfall listener registered `prepend: true` that post-processes `await next()` — **final-assembly enforcement**: the assembly the loop renders never carries `structured_output` for an agent without a structured run, and for one that has it always carries the run's OWN schema (as the tool's `parameters`) plus the calling instruction as a trailing prompt section (the demand travels with the tool — `AgentOptions` has no per-agent prompt field to carry it). The loop logs the rendered assembly as the step's `request/header`, so the injection is reconstructable log state, never a wire-only mutation. Per-agent shaping lives here because the tool registry and prompt assembly are context-global while schemas differ per concurrent child (FIXME in the module doc: per-agent/per-session scoping would dissolve this); cooperative mutate-then-`next()` would not survive a downstream listener returning a replacement assembly. +- a `tools/post-execute` listener (`prepend: true` = outermost, so `await next()` yields the composed final decision) that COMMITS the capture: the tool body only stages the validated value, and it becomes the run's result only when the final decision accepts the call — a downstream block (a PostToolUse hook) turns the logged result into `isError`, and the run must not report `structured` success for a call the model and session log saw fail. +- a `tools/pre-execute` deny for any call arriving after the agent's capture — terminal means terminal WITHIN the step: a response listing `structured_output` before further tool calls cannot run side effects after the final answer was accepted. - an `agent/turn-continuation` listener (also `prepend: true` — an earlier-registered force-continue listener returning without `next()` must not decide the turn before the veto runs) that stops a child's turn once its output is captured, so a successful capture doesn't buy a wasted extra model step. -The capture tool validates each call against the run's schema (`validateStructuredValue`) — violations become an `INVALID_ARGS` isError result the model retries in-turn; a valid call records the value. +The capture tool validates each call against the run's schema (`validateStructuredValue`) — violations become an `INVALID_ARGS` isError result the model retries in-turn; a valid call stages the value for the post-execute commit. -Lifetime is refcounted with two kinds of holder: each backend acquires for its plugin lifetime (`apply`), and each structured RUN holds its own acquisition from start to settle — so unregistration can never precede a live run's settle, and the runtime disposes only when the last backend AND the last run are gone. `release()` is idempotent per acquisition. +Lifetime is refcounted by live structured runs: each acquires at start and releases at settle, so the registrations exist exactly while at least one structured child is live, a backend hot-reload mid-run cannot unregister the capture tool under a live child, and the last settle disposes everything. `release()` is idempotent per acquisition. ### `depthOf(agent): number` diff --git a/packages/subagent/subagent-inprocess/package.json b/packages/subagent/subagent-inprocess/package.json index ecc177162f..4e6b72533a 100644 --- a/packages/subagent/subagent-inprocess/package.json +++ b/packages/subagent/subagent-inprocess/package.json @@ -37,8 +37,6 @@ "@deepseek-ai/dsh-llm": "workspace:^", "@deepseek-ai/dsh-session": "workspace:^", "@deepseek-ai/dsh-subagent": "workspace:^", - "@deepseek-ai/dsh-subagent-fork": "workspace:^", - "@deepseek-ai/dsh-subagent-spawn": "workspace:^", "@deepseek-ai/dsh-system-prompt": "workspace:^", "@deepseek-ai/dsh-tools": "workspace:^", "cordis": "^4.0.0-rc.6" diff --git a/packages/subagent/subagent-spawn/README.md b/packages/subagent/subagent-spawn/README.md index 696007a693..b976ef5a63 100644 --- a/packages/subagent/subagent-spawn/README.md +++ b/packages/subagent/subagent-spawn/README.md @@ -10,7 +10,7 @@ The run mechanics live in the shared [`@deepseek-ai/dsh-subagent-inprocess`](../ ## Capabilities -`{ outputSchema: true, depthLimit: true, toolFilter: false }`. It constructs the child, so it enforces a recursion cap, and it supports structured output via the driver's shared [structured runtime](../subagent-inprocess/README.md) (the backend acquires it for its plugin lifetime; each structured run holds its own acquisition until it settles). Tool-scoping is deferred (the service rejects a request needing it before `start` runs). +`{ outputSchema: true, depthLimit: true, toolFilter: false }`. It constructs the child, so it enforces a recursion cap, and it supports structured output via the driver's [structured runtime](../subagent-inprocess/README.md) (acquired per structured run inside the driver — this backend registers nothing at apply). Tool-scoping is deferred (the service rejects a request needing it before `start` runs). ## Config diff --git a/packages/subagent/subagent-spawn/src/index.ts b/packages/subagent/subagent-spawn/src/index.ts index 7da954bc44..2d8f118b4e 100644 --- a/packages/subagent/subagent-spawn/src/index.ts +++ b/packages/subagent/subagent-spawn/src/index.ts @@ -22,14 +22,14 @@ import type { Context } from 'cordis' import z from 'schemastery' import type { SubagentCapabilities, SubagentProvider, SubagentStartRequest } from '@deepseek-ai/dsh-subagent' -import { acquireStructuredRuntime, startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' +import { startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess' export const name = 'subagent-spawn' -// `tools` is deliberately NOT injected: the structured runtime gates its own -// capture-tool registration on `tools` availability internally, so this -// backend's apply timing — and with it the provider-mirroring delegation -// tool's position in the model-visible tool list — stays what it was before -// structured output existed. +// `tools` is deliberately NOT injected: the shared driver's structured runtime +// (acquired per structured RUN, not at apply) gates its own capture-tool +// registration on `tools` availability, so this backend's apply timing — and +// with it the provider-mirroring delegation tool's position in the +// model-visible tool list — stays what it was before structured output existed. export const inject = ['subagents', 'agents'] /** Config: the registry name to register the provider under. */ @@ -64,13 +64,5 @@ class SpawnProvider implements SubagentProvider { } export function apply(ctx: Context, config: Config): void { - // Hold the structured runtime for the plugin's lifetime, so the capture tool - // and its request-shaping listeners are registered before the first - // structured run and torn down when the last backend unloads (live runs hold - // their own acquisitions, so an unload mid-run cannot strand a child). - ctx.effect(() => { - const acquisition = acquireStructuredRuntime(ctx) - return () => { acquisition.release() } - }, 'subagent-spawn structured runtime') ctx.subagents.registerProvider(new SpawnProvider(config.providerName, ctx)) } diff --git a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts index 63d26c0531..1dad9748e9 100644 --- a/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts +++ b/packages/subagent/subagent-spawn/tests/subagent-spawn.spec.ts @@ -10,9 +10,9 @@ import { SessionId } from '@deepseek-ai/dsh-session' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import * as Invariants from '@deepseek-ai/dsh-invariants' import SubagentService from '@deepseek-ai/dsh-subagent' -import { MockAdapter, maxTokensResponse, textResponse } from '../../../core/agent-loop/tests/mock-adapter.ts' +import { MockAdapter, maxTokensResponse, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts' import * as spawn from '../src/index.ts' -import { depthOf, SubagentDepthError } from '@deepseek-ai/dsh-subagent-inprocess' +import { depthOf, STRUCTURED_OUTPUT_TOOL, SubagentDepthError } from '@deepseek-ai/dsh-subagent-inprocess' type Script = ConstructorParameters[0] @@ -251,18 +251,60 @@ describe('dsh-subagent-spawn', () => { const ctx = new Context() await ctx.plugin(SubagentService) await ctx.plugin(AgentRegistry) - // The backend does NOT inject 'tools' (the structured runtime gates its - // capture-tool registration on tools availability itself, keeping backend - // apply timing — and the delegation tool's prompt position — unchanged); - // the registries are loaded here so the runtime registers eagerly anyway. - await ctx.plugin(SystemPrompt) - await ctx.plugin(ToolRegistry) const fiber = await ctx.plugin(spawn, { providerName: 'spawn' }) expect(ctx.subagents.list()).toEqual(['spawn']) await fiber.dispose() expect(ctx.subagents.list()).toEqual([]) }) + it('captures structured output through the shipped plugin (driver runtime, plugin wiring)', async () => { + const { ctx, parent } = await setup([ + toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }), + ]) + const run = ctx.subagents.start('spawn', { + prompt: [{ type: 'text', text: 'produce the answer' }], + parent, + outputSchema: { type: 'object', properties: { answer: { type: 'number' } }, required: ['answer'] }, + }) + const result = await run.result + expect(result.stopReason).toBe('completed') + expect(result.structured).toEqual({ answer: 42 }) + // Run-scoped runtime: the settle released the last acquisition. + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + await run.dispose() + }) + + it('a backend unload mid-structured-run settles the run and releases the runtime', async () => { + // Rebuild the stack by hand so we hold the backend's fiber. + const ctx = new Context() + const adapter = new MockAdapter(['hang']) + await ctx.plugin(LlmService) + await ctx.plugin(SessionStore) + await ctx.plugin(SystemPrompt) + await ctx.plugin(ToolRegistry) + await ctx.plugin(AgentRegistry) + await ctx.plugin(Invariants) + await ctx.plugin(AgentLoop, { agents: [] }) + await ctx.plugin(SubagentService) + const fiber = await ctx.plugin(spawn, { providerName: 'spawn' }) + ctx.llm.registerAdapter(['mock'], adapter) + const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' }) + const run = ctx.subagents.start('spawn', { + prompt: [{ type: 'text', text: 'q' }], + parent, + outputSchema: { type: 'object', properties: { a: { type: 'number' } } }, + }) + // Let the child's step start streaming, then unload the backend. The + // backend owns the child agent, so the unload tears the child down and + // the run settles — releasing its own runtime acquisition on the way out. + await new Promise(resolve => setTimeout(resolve, 30)) + await fiber.dispose() + const result = await run.result + expect(result.stopReason).toBe('error') + expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined() + await run.dispose() + }) + it('has the namespace-plugin export shape (no stray default)', () => { expect('default' in spawn).toBe(false) expect(spawn.name).toBe('subagent-spawn') diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 35d940028b..d5fb68c746 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -655,12 +655,6 @@ importers: '@deepseek-ai/dsh-subagent': specifier: workspace:^ version: link:../subagent - '@deepseek-ai/dsh-subagent-fork': - specifier: workspace:^ - version: link:../subagent-fork - '@deepseek-ai/dsh-subagent-spawn': - specifier: workspace:^ - version: link:../subagent-spawn '@deepseek-ai/dsh-system-prompt': specifier: workspace:^ version: link:../../core/system-prompt From 23fb1febcd585b2cf32bf0eff28244dc73229644 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 21:17:52 +0800 Subject: [PATCH 16/24] chore: keep the type-equiv manifest in its one-line-per-entry format The previous commit rewrote the whole file through a JSON pretty-printer, reformatting every existing entry; restore the established compact style with the four new entries appended to the tools.md group. --- scripts/type-equiv.manifest.json | 453 ++++++------------------------- 1 file changed, 83 insertions(+), 370 deletions(-) diff --git a/scripts/type-equiv.manifest.json b/scripts/type-equiv.manifest.json index e414cff1f4..b66882d8c9 100644 --- a/scripts/type-equiv.manifest.json +++ b/scripts/type-equiv.manifest.json @@ -1,375 +1,88 @@ { "comment": "Maps each ` ```ts type-equiv ` block (by doc + declared symbol) to the source symbol it must match verbatim. verify-type-equiv.ts enforces a 1:1 correspondence: every type-equiv block has exactly one entry here, and every entry resolves to exactly one block. Add an entry when you add a type-equiv block; remove it when you remove the block.", "entries": [ - { - "doc": "docs/core-data-structures/core.md", - "symbol": "Branded", - "source": "packages/util/brand/src/index.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "ContentBlockMap", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "Message", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "MessageSourceMap", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "FinishReasonMap", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "GenerateOptions", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "ToolSchema", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "LlmCallConfig", - "source": "packages/llm/llm/src/call-config.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "SessionEvent", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "Agent", - "source": "packages/core/agent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "HookContext", - "source": "packages/core/agent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "PromptDecision", - "source": "packages/core/agent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "ContinuationDecision", - "source": "packages/core/agent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/core.md", - "symbol": "SessionStartSource", - "source": "packages/core/agent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/llm-streaming.md", - "symbol": "StreamChunk", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/llm-streaming.md", - "symbol": "TokenUsage", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/llm-streaming.md", - "symbol": "ContentBlockMap", - "source": "packages/llm/llm/src/types.ts" - }, - { - "doc": "docs/core-data-structures/llm-streaming.md", - "symbol": "AppIdentity", - "source": "packages/llm/llm/src/attribution.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "SessionEventMap", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "EpochHeader", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "TodoItem", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "SessionEvent", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "TurnTriggerMap", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "TurnEndReasonMap", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "SurfaceEventType", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "SurfaceOp", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "SurfaceIntent", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/session.md", - "symbol": "SurfaceNode", - "source": "packages/core/session/src/surface.ts" - }, - { - "doc": "docs/core-data-structures/persistence.md", - "symbol": "SessionHeader", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/persistence.md", - "symbol": "CreateSessionOptions", - "source": "packages/core/session/src/types.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "ToolDefinition", - "source": "packages/core/tools/src/index.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "SchemaProp", - "source": "packages/core/tools/src/schema.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "SchemaSpec", - "source": "packages/core/tools/src/schema.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "InferArgs", - "source": "packages/core/tools/src/schema.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "ToolExecution", - "source": "packages/core/tools/src/index.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "ToolExecutionResult", - "source": "packages/core/tools/src/index.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "PreToolDecision", - "source": "packages/core/tools/src/index.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "PostToolDecision", - "source": "packages/core/tools/src/index.ts" - }, - { - "doc": "docs/core-data-structures/bash.md", - "symbol": "BashExecRequest", - "source": "packages/bash/bash/src/types.ts" - }, - { - "doc": "docs/core-data-structures/bash.md", - "symbol": "BashExecSpec", - "source": "packages/bash/bash/src/types.ts" - }, - { - "doc": "docs/core-data-structures/bash.md", - "symbol": "BashRunResult", - "source": "packages/bash/bash/src/types.ts" - }, - { - "doc": "docs/core-data-structures/bash.md", - "symbol": "CollectedOutput", - "source": "packages/bash/bash/src/types.ts" - }, - { - "doc": "docs/core-data-structures/bash.md", - "symbol": "BashTask", - "source": "packages/bash/bash/src/types.ts" - }, - { - "doc": "docs/core-data-structures/bash.md", - "symbol": "BashTaskRead", - "source": "packages/bash/bash/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsTarget", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsTargetKey", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsVersion", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsInfo", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsDirEntry", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsWriteIntent", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsWriteOutcome", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsEditRequest", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsEditOutcome", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsErrorCode", - "source": "packages/fs/fs/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FsPolicyExec", - "source": "packages/fs/fs-policy/src/types.ts" - }, - { - "doc": "docs/core-data-structures/filesystem.md", - "symbol": "FileReadOutcome", - "source": "packages/fs/tool-fs/src/read-render.ts" - }, - { - "doc": "docs/core-data-structures/compaction.md", - "symbol": "CompactionResult", - "source": "packages/compact/compact/src/types.ts" - }, - { - "doc": "docs/core-data-structures/subagent.md", - "symbol": "SubagentCapabilities", - "source": "packages/subagent/subagent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/subagent.md", - "symbol": "SubagentStartRequest", - "source": "packages/subagent/subagent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/subagent.md", - "symbol": "SubagentResult", - "source": "packages/subagent/subagent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/subagent.md", - "symbol": "SubagentStopReasonMap", - "source": "packages/subagent/subagent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/subagent.md", - "symbol": "SubagentRun", - "source": "packages/subagent/subagent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/subagent.md", - "symbol": "SubagentProvider", - "source": "packages/subagent/subagent/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebSearchRequest", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebSearchResult", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebSearchSource", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebFetchRequest", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebFetchResult", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebFetchBody", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/web.md", - "symbol": "WebProviderStatus", - "source": "packages/web/web/src/types.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "StructuredScalar", - "source": "packages/core/tools/src/json-schema.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "StructuredSchemaType", - "source": "packages/core/tools/src/json-schema.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "StructuredSchemaNode", - "source": "packages/core/tools/src/json-schema.ts" - }, - { - "doc": "docs/core-data-structures/tools.md", - "symbol": "StructuredOutputSchema", - "source": "packages/core/tools/src/json-schema.ts" - } + { "doc": "docs/core-data-structures/core.md", "symbol": "Branded", "source": "packages/util/brand/src/index.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "ContentBlockMap", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "Message", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "MessageSourceMap", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "FinishReasonMap", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "GenerateOptions", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "ToolSchema", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "LlmCallConfig", "source": "packages/llm/llm/src/call-config.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "Agent", "source": "packages/core/agent/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "HookContext", "source": "packages/core/agent/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "PromptDecision", "source": "packages/core/agent/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "ContinuationDecision", "source": "packages/core/agent/src/types.ts" }, + { "doc": "docs/core-data-structures/core.md", "symbol": "SessionStartSource", "source": "packages/core/agent/src/types.ts" }, + + { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "StreamChunk", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "TokenUsage", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "ContentBlockMap", "source": "packages/llm/llm/src/types.ts" }, + { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "AppIdentity", "source": "packages/llm/llm/src/attribution.ts" }, + + { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEventMap", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "EpochHeader", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "TodoItem", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "TurnTriggerMap", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "TurnEndReasonMap", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceEventType", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceOp", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceIntent", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/session.md", "symbol": "SurfaceNode", "source": "packages/core/session/src/surface.ts" }, + + { "doc": "docs/core-data-structures/persistence.md", "symbol": "SessionHeader", "source": "packages/core/session/src/types.ts" }, + { "doc": "docs/core-data-structures/persistence.md", "symbol": "CreateSessionOptions", "source": "packages/core/session/src/types.ts" }, + + { "doc": "docs/core-data-structures/tools.md", "symbol": "ToolDefinition", "source": "packages/core/tools/src/index.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "SchemaProp", "source": "packages/core/tools/src/schema.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "SchemaSpec", "source": "packages/core/tools/src/schema.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "InferArgs", "source": "packages/core/tools/src/schema.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "ToolExecution", "source": "packages/core/tools/src/index.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "ToolExecutionResult", "source": "packages/core/tools/src/index.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "PreToolDecision", "source": "packages/core/tools/src/index.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "PostToolDecision", "source": "packages/core/tools/src/index.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "StructuredScalar", "source": "packages/core/tools/src/json-schema.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "StructuredSchemaType", "source": "packages/core/tools/src/json-schema.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "StructuredSchemaNode", "source": "packages/core/tools/src/json-schema.ts" }, + { "doc": "docs/core-data-structures/tools.md", "symbol": "StructuredOutputSchema", "source": "packages/core/tools/src/json-schema.ts" }, + + { "doc": "docs/core-data-structures/bash.md", "symbol": "BashExecRequest", "source": "packages/bash/bash/src/types.ts" }, + { "doc": "docs/core-data-structures/bash.md", "symbol": "BashExecSpec", "source": "packages/bash/bash/src/types.ts" }, + { "doc": "docs/core-data-structures/bash.md", "symbol": "BashRunResult", "source": "packages/bash/bash/src/types.ts" }, + { "doc": "docs/core-data-structures/bash.md", "symbol": "CollectedOutput", "source": "packages/bash/bash/src/types.ts" }, + { "doc": "docs/core-data-structures/bash.md", "symbol": "BashTask", "source": "packages/bash/bash/src/types.ts" }, + { "doc": "docs/core-data-structures/bash.md", "symbol": "BashTaskRead", "source": "packages/bash/bash/src/types.ts" }, + + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsTarget", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsTargetKey", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsVersion", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsInfo", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsDirEntry", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsWriteIntent", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsWriteOutcome", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsEditRequest", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsEditOutcome", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsErrorCode", "source": "packages/fs/fs/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FsPolicyExec", "source": "packages/fs/fs-policy/src/types.ts" }, + { "doc": "docs/core-data-structures/filesystem.md", "symbol": "FileReadOutcome", "source": "packages/fs/tool-fs/src/read-render.ts" }, + + { "doc": "docs/core-data-structures/compaction.md", "symbol": "CompactionResult", "source": "packages/compact/compact/src/types.ts" }, + + { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentCapabilities", "source": "packages/subagent/subagent/src/types.ts" }, + { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentStartRequest", "source": "packages/subagent/subagent/src/types.ts" }, + { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentResult", "source": "packages/subagent/subagent/src/types.ts" }, + { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentStopReasonMap", "source": "packages/subagent/subagent/src/types.ts" }, + { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentRun", "source": "packages/subagent/subagent/src/types.ts" }, + { "doc": "docs/core-data-structures/subagent.md", "symbol": "SubagentProvider", "source": "packages/subagent/subagent/src/types.ts" }, + + { "doc": "docs/core-data-structures/web.md", "symbol": "WebSearchRequest", "source": "packages/web/web/src/types.ts" }, + { "doc": "docs/core-data-structures/web.md", "symbol": "WebSearchResult", "source": "packages/web/web/src/types.ts" }, + { "doc": "docs/core-data-structures/web.md", "symbol": "WebSearchSource", "source": "packages/web/web/src/types.ts" }, + { "doc": "docs/core-data-structures/web.md", "symbol": "WebFetchRequest", "source": "packages/web/web/src/types.ts" }, + { "doc": "docs/core-data-structures/web.md", "symbol": "WebFetchResult", "source": "packages/web/web/src/types.ts" }, + { "doc": "docs/core-data-structures/web.md", "symbol": "WebFetchBody", "source": "packages/web/web/src/types.ts" }, + { "doc": "docs/core-data-structures/web.md", "symbol": "WebProviderStatus", "source": "packages/web/web/src/types.ts" } ] } From aeaccf6d360e9475b54adfd61096a9cef1e0ab25 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Tue, 7 Jul 2026 22:13:01 +0800 Subject: [PATCH 17/24] docs: satisfy the new export-JSDoc gate on the assertion signature Master's verify-export-jsdoc (landed mid-stack) wants @returns on every exported function including asserts-returning ones; document the narrowing. --- packages/core/tools/src/json-schema.ts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/packages/core/tools/src/json-schema.ts b/packages/core/tools/src/json-schema.ts index bc0da537e1..4c1036773b 100644 --- a/packages/core/tools/src/json-schema.ts +++ b/packages/core/tools/src/json-schema.ts @@ -256,6 +256,8 @@ function checkSchemaNode(node: unknown, path: string, violations: string[], seen * (`UNSUPPORTED_SCHEMA`) listing EVERY violation; returns (and narrows) on * success. Call this at the seam boundary, before any child is created. * @param schema - the caller-supplied schema (unknown until asserted). + * @returns nothing — the assertion signature narrows `schema` to + * {@link StructuredOutputSchema} in the caller's scope on normal return. */ export function assertSupportedOutputSchema(schema: unknown): asserts schema is StructuredOutputSchema { const violations: string[] = [] From 020595529486c79e1cced060eb9adfe95f5e086f Mon Sep 17 00:00:00 2001 From: imccyu <276526105+imccyu@users.noreply.github.com> Date: Tue, 7 Jul 2026 23:21:10 +0800 Subject: [PATCH 18/24] docs: update budget --- scripts/doc-budgets.manifest.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/doc-budgets.manifest.json b/scripts/doc-budgets.manifest.json index 94afc8dc73..b23811e3d0 100644 --- a/scripts/doc-budgets.manifest.json +++ b/scripts/doc-budgets.manifest.json @@ -1,5 +1,5 @@ { - "AGENTS.md": 1690, + "AGENTS.md": 1691, "docs/AGENTS.md": 1315, "docs/architecture.md": 1630, "docs/cordis-primer.md": 550, From 0f021d8efc644dd7c736ad473e0f45f8e4d758be Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 01:25:43 +0800 Subject: [PATCH 19/24] rfc(testing): propose extracting the ACP snapshot suite into a support package MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The harness, normalizers, and suite/guard logic live inside examples/acp-agent/tests, outside the coverage gate and copyable-only for a second suite. Propose @deepseek-ai/dsh-acp-snapshot under packages/support: parameterized runScenario, verbatim normalizers, a defineAcpSnapshotSuite factory with per-suite header pinning, and scripted permissionAnswers so an approval round-trip is expressible at the snapshot tier — the sandbox composition is the immediate consumer. --- docs/rfc/INDEX.md | 1 + .../2026-07-08-shared-acp-snapshot-package.md | 54 +++++++++++++++++++ 2 files changed, 55 insertions(+) create mode 100644 docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index 7b78a08690..39927e3095 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -42,6 +42,7 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; |---|---| | [Deterministic tests, the replay invariant fixture, and race stress](proposed/testing/2026-06-11-deterministic-and-stress-testing.md) | 2026-06-11 | | [Mutation testing as the coverage counterweight](proposed/testing/2026-06-11-mutation-testing.md) | 2026-06-11 | +| [Extract the ACP snapshot suite into a support package](proposed/testing/2026-07-08-shared-acp-snapshot-package.md) | 2026-07-08 | ## Implemented diff --git a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md new file mode 100644 index 0000000000..2bd061d422 --- /dev/null +++ b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md @@ -0,0 +1,54 @@ +# RFC: Extract the ACP snapshot suite into a support package + +Status: proposed + +## Problem + +The ACP snapshot tier ([snapshot RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md)) is built from three modules that live inside one example's test directory: [snapshot-harness.ts](../../../../examples/acp-agent/tests/snapshot-harness.ts) (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), [snapshot-normalize.ts](../../../../examples/acp-agent/tests/snapshot-normalize.ts) (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in [acp.snapshot.ts](../../../../examples/acp-agent/tests/acp.snapshot.ts) (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). + +A second ACP example that wants snapshot coverage — the sandbox/approval composition is the immediate consumer — can only copy those modules, forking exactly the logic that must not drift: record write-back, header scrubbing, child-session harvest ordering. The spawn/client glue is already triplicated across [acp.e2e.ts](../../../../examples/acp-agent/tests/acp.e2e.ts), [hooks.e2e.ts](../../../../examples/acp-agent/tests/hooks.e2e.ts), and the harness, marked by `TODO(acp-test-harness)`. + +Location also decides test rigor: the per-file 100% coverage gate measures `packages/*/*/src` only, so none of this machinery is measured — the same gap that moved `dsh-llm-replay` out of `examples/` into [packages/support](../../../../packages/support/README.md). The harness's subprocess lifecycle, teardown, and harvest-ordering branches are exercised only transitively, when a live scenario happens to hit them. + +Finally, the harness's ACP client hardcodes `requestPermission → cancelled`, so an approval round-trip — the headline behavior of the sandbox composition — cannot be expressed at the snapshot tier at all. A new transcript surface must name its coverage at every tier at plan time; today the tier cannot express this one. + +## Proposal + +Create `packages/support/acp-snapshot` (`@deepseek-ai/dsh-acp-snapshot`), a support-tier package with three source modules; each example keeps only its scenario table, its `snapshots/` fixtures, its `cordis.snapshot.yml` overlay ([single-source replay config](../../implemented/testing/2026-07-04-single-source-acp-replay-config.md)), and the paths that identify its agent. + +**`src/harness.ts`** — `runScenario` and the input-script/result types, moved intact, with the module-level path constants replaced by an explicit `AgentUnderTest` parameter (`binScript`, `configPath`, `tsconfigPath`): defaulting stays at the seam's consumer, which resolves them from its own `import.meta.url`. The internal spawn/tee/SDK-client wiring is factored so the e2e launcher duplication can migrate onto it later; that migration is out of scope here and the `TODO(acp-test-harness)` stays until it lands. + +**`src/normalize.ts`** — the normalizers move verbatim with their spec. They stay hook-free: when a future event carries a new volatile field (an approval duration, say), the shared normalizer learns it in the same change, keeping one home for what "normalized" means rather than per-suite scrub extensions. + +**`src/suite.ts`** — the `Scenario` type and `defineAcpSnapshotSuite(options)`, which registers the per-scenario `describe`/`it` tree and the fixture guard tests. Options carry the resolved `mode: 'replay' | 'record'` — reading `DSH_SNAPSHOT` stays at the edge, in the example's `*.snapshot.ts`. The guard logic (no orphan scenario dirs, required fixture files, exactly one pin, non-pinning fixtures are `scrubRequestHeaders` fixed points) is exported as pure assertion functions the registered tests call one-line-each, so failure paths are unit-testable without meta-running vitest. The pinned-header contract ([pinned-header RFC](../../implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)) becomes per-suite: each suite flags exactly one `pinsHeader` scenario and its uniformity guard compares only that suite's sessions, which is the guard's existing scope. + +**Scripted permission answers** — `InputScript` gains an optional ordered `permissionAnswers` queue consumed by the harness client's `requestPermission`, each entry selecting a response by option **kind** (`allow_once`, `reject_once`, …); the harness maps kind to the agent-issued `optionId` at answer time, since ids are random per run while kinds are stable. An exhausted or absent queue falls back to today's `cancelled`, so existing scenarios and goldens are untouched. This is what lets a sandbox suite script an approval round-trip deterministically from `input.json`. + +Repo wiring follows the [adding-a-package cookbook](../../../cookbook/adding-a-package.md): manifest per the workspace constraints (cordis peer+dev, `private`, standard `files`), references in the root and build tsconfigs (the `@deepseek-ai/dsh-*` paths wildcard already covers `packages/support/*/src`), a row in the support group README, and explicit `@agentclientprotocol/sdk`/`vitest`/`tsx` dependencies instead of inherited-by-walk-up resolution. [docs/testing.md](../../../testing.md) generalizes "scenarios live under `examples/acp-agent/tests/snapshots/`" to the owning example's `tests/snapshots/`. + +Landing order is three commits on one PR: (1) the pure move plus parameterization, with `examples/acp-agent/tests/acp.snapshot.ts` collapsed to its scenario table and one `defineAcpSnapshotSuite` call; (2) the coverage work — a scripted fake ACP bin fixture (reads JSON-RPC frames on stdin, emits canned responses and session updates, writes synthetic session JSONL under `DSH_SNAPSHOT_SESSIONS_ROOT`) driving `harness.ts` through every step op, expect-error branch, child-harvest ordering, and teardown path, and `suite.spec.ts` registering synthetic replay- and record-mode suites against temp fixture dirs (record mode is keyless here: the live API sits behind the bin, and the fake bin needs none); (3) `permissionAnswers` with its unit coverage. The sandbox branch then merges master down and adds its own suite: scenario table, own pin scenario, own overlay, fixtures recorded via `test:snapshot:record`. + +## Alternatives considered + +- **Copy the modules into each example** — the fork this RFC exists to prevent: the record/guard logic is exactly the code that must stay byte-identical across suites, and examples are outside the coverage gate, so each copy is also unmeasured. +- **A shared module directory under `examples/`** — keeps the code outside the coverage gate and forces relative imports across example boundaries, against the package-name import convention; `examples/` leaves stay thin by design. +- **A `/testing` subpath export of `dsh-acp-agent`** — couples test infrastructure into a product package's surface and dependency set; `packages/support/` exists precisely for real-but-lower-compatibility dev/test packages, with `dsh-llm-replay` as the precedent this proposal completes. +- **Export raw test-body functions instead of a suite factory** — each example would re-own the `describe`/`it` skeleton (~80 lines of registration boilerplate per suite) for no flexibility gain; the factory keeps consumers to a scenario table plus one call, and the pure guard functions preserve unit-testability inside the factory design. +- **An injectable ACP `Client` factory instead of declarative `permissionAnswers`** — maximally flexible, but it leaks SDK client construction to every consumer and reopens per-example drift in exactly the layer being unified; a declarative queue keeps `input.json` the single scripting surface and stays golden-normalizable. +- **Generalize beyond ACP (a transport-agnostic snapshot harness)** — no second transport exists; the harness is ACP-shaped end to end (SDK client, JSON-RPC frames, `session/update` waiters), and a speculative abstraction would be a seam split ahead of any consumer. + +## Acceptance criteria + +- After the pure-move commit, `pnpm run test:snapshot` is green with zero byte changes under `examples/acp-agent/tests/snapshots/` — the machine proof that extraction changed no behavior. +- `pnpm run test:coverage` holds the new package's `src/` at per-file 100% with keyless unit specs; any `v8 ignore` carries its reason. +- `examples/acp-agent/tests/acp.snapshot.ts` contains no golden/compare/guard logic — only the scenario table, the agent paths, and the factory call. +- A harness unit test drives a `permissionAnswers` script through kind→`optionId` mapping and the exhaustion fallback, demonstrating the tier can express an approval round-trip before the sandbox suite needs it. +- `doc-sync`, `hygiene`, and `verify-module-graph` pass with the new package wired in. + +## Risks + +- **Per-file 100% on `suite.ts`** is the tightest constraint: `toMatchFileSnapshot` update semantics differ under CI, and factory-registered tests must be driven by real vitest collection. The pure-guard-function split plus synthetic-suite registration is the mitigation; a justified `v8 ignore` is the last resort, not the plan. +- **`vitest` becomes a `src` dependency** of a workspace package (the factory imports `describe`/`it`/`expect`), so importing `suite.ts` outside a vitest run throws — acceptable for a support-tier package and stated in its README, but it is a shape no other package has. +- **Fake-bin drift**: harness unit tests exercise plumbing against a scripted bin, not the real one. The real bin path stays exercised on every `test:snapshot` run, so drift surfaces there; the fake bin only owns branches the live suite cannot deterministically reach. +- **Per-suite pins duplicate header bulk**: each new suite commits one full ~8 KB header fixture. Accepted — a suite whose composition equals another's is the degenerate case the uniformity guard would surface, and one pinned line per genuinely distinct composition is the pinned-header design applied at its natural scope. +- The extraction touches the gating snapshot suite itself; a subtle behavior change would surface as golden churn. The zero-byte-diff acceptance criterion is the guard. From 556f8470642218ad2235d73efe0511f1e1587289 Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 01:44:20 +0800 Subject: [PATCH 20/24] feat(acp-snapshot): extract the ACP snapshot suite into a support package MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The snapshot tier's machinery leaves examples/acp-agent/tests for packages/support/acp-snapshot (@deepseek-ai/dsh-acp-snapshot), where the coverage gate measures it and a second example can consume it instead of forking it: harness.ts (runScenario, parameterized by an AgentUnderTest {binScript, configPath, tsconfigPath} instead of module constants), normalize.ts (moved verbatim), and suite.ts (defineAcpSnapshotSuite — the per-scenario golden/log compares, record write-back, per-suite header pin with its uniformity guard, and the fixture guard block, lifted from acp.snapshot.ts). The example file collapses to its scenario table plus one factory call; env reading (DSH_SNAPSHOT) stays at that edge. The exactly-one-pin meta-test generalizes from the hardcoded text-turn name to "exactly one per suite" — which scenario pins is the scenario table's reviewable choice (per-suite pinning per the proposal RFC). Extraction parity: pnpm run test:snapshot is 36 passed + fs-policy-reject failing BEFORE AND AFTER (BSD-sed environment failure, reproduced at the base commit in a clean worktree — the recorded golden's sed -i syntax is GNU-only), with zero byte changes under examples/acp-agent/tests/snapshots/. Coverage for the new src files lands in the next commit. --- docs/config-catalog.md | 1 + docs/module-graph.md | 2 + ...0-remove-redundant-snapshot-log-goldens.md | 2 +- ...-request-header-content-in-one-scenario.md | 2 +- .../2026-07-08-shared-acp-snapshot-package.md | 2 +- docs/testing.md | 2 +- examples/acp-agent/tests/acp.e2e.ts | 4 +- examples/acp-agent/tests/acp.snapshot.ts | 331 +--------------- knip.json | 7 +- packages/support/README.md | 3 +- packages/support/acp-snapshot/README.md | 36 ++ packages/support/acp-snapshot/package.json | 35 ++ .../support/acp-snapshot/src/harness.ts | 79 ++-- packages/support/acp-snapshot/src/index.ts | 37 ++ .../support/acp-snapshot/src/normalize.ts | 22 +- packages/support/acp-snapshot/src/suite.ts | 355 ++++++++++++++++++ .../acp-snapshot/tests/normalize.spec.ts | 4 +- packages/support/acp-snapshot/tsconfig.json | 11 + pnpm-lock.yaml | 68 ++++ tsconfig.build.json | 1 + tsconfig.json | 1 + 21 files changed, 654 insertions(+), 351 deletions(-) create mode 100644 packages/support/acp-snapshot/README.md create mode 100644 packages/support/acp-snapshot/package.json rename examples/acp-agent/tests/snapshot-harness.ts => packages/support/acp-snapshot/src/harness.ts (86%) create mode 100644 packages/support/acp-snapshot/src/index.ts rename examples/acp-agent/tests/snapshot-normalize.ts => packages/support/acp-snapshot/src/normalize.ts (91%) create mode 100644 packages/support/acp-snapshot/src/suite.ts rename examples/acp-agent/tests/snapshot-normalize.spec.ts => packages/support/acp-snapshot/tests/normalize.spec.ts (98%) create mode 100644 packages/support/acp-snapshot/tsconfig.json diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 2c101e318e..e268f63da6 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -810,6 +810,7 @@ Abstract service classes — a deployment loads a concrete implementation packag Imported as libraries by other packages; a `cordis.yml` cannot load them. +- `@deepseek-ai/dsh-acp-snapshot` ([`packages/support/acp-snapshot/src/index.ts`](../packages/support/acp-snapshot/src/index.ts)) - `@deepseek-ai/dsh-app-boot` ([`packages/ui/app-boot/src/index.ts`](../packages/ui/app-boot/src/index.ts)) - `@deepseek-ai/dsh-brand` ([`packages/util/brand/src/index.ts`](../packages/util/brand/src/index.ts)) - `@deepseek-ai/dsh-hook-protocol` ([`packages/hooks/hook-protocol/src/index.ts`](../packages/hooks/hook-protocol/src/index.ts)) diff --git a/docs/module-graph.md b/docs/module-graph.md index 8283a8b756..c08b1fd171 100644 --- a/docs/module-graph.md +++ b/docs/module-graph.md @@ -68,6 +68,7 @@ flowchart TD pkg_session_persistence_sqlite["session-persistence-sqlite"] end subgraph group_support["packages/support"] + pkg_acp_snapshot["acp-snapshot"] pkg_invariants["invariants"] pkg_llm_replay["llm-replay"] pkg_subagent_mock["subagent-mock"] @@ -207,6 +208,7 @@ flowchart TD | Package | Group | Depends on | | --- | --- | --- | | [`brand`](../packages/util/brand) | `util` | — | +| [`acp-snapshot`](../packages/support/acp-snapshot) | `support` | — | | [`app-boot`](../packages/ui/app-boot) | `ui` | — | | [`llm`](../packages/llm/llm) | `llm` | [`brand`](../packages/util/brand) | | [`bash`](../packages/bash/bash) | `bash` | [`brand`](../packages/util/brand) | diff --git a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md index 07a1cfd731..2a6f8b7ae3 100644 --- a/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md +++ b/docs/rfc/implemented/testing/2026-06-20-remove-redundant-snapshot-log-goldens.md @@ -32,4 +32,4 @@ Reviewers lose one artifact name that made the expected persisted log visually s ## Implementation note -The comparison normalizes BOTH sides, but each against its OWN volatile values, not a shared context. A raw harvested `session.jsonl` bakes in the recording run's session id, cwd, and timestamps; the replay run produces fresh ones. `normalizeSessionLog` scrubs cwd by exact string match, so normalizing the fixture against the *replay* run's cwd would leave the recorded cwd in the header unscrubbed and the compare would fail. The harness therefore derives the fixture's normalize context from its OWN header line (`{ type:'session', id, cwd }`) — `fixtureContext()` in `acp.snapshot.ts` — so both sides scrub to the same `{{sessionId}}`/`{{cwd}}` tokens. An authored fixture copied from the old golden already carries the normalized header (`id:'{{sessionId}}'`, `cwd:'{{cwd}}'`), which yields those tokens as the volatile values and scrubs idempotently. The session-log side uses a plain normalized-string `toEqual`, NOT `toMatchFileSnapshot`, so a run never overwrites the fixture. +The comparison normalizes BOTH sides, but each against its OWN volatile values, not a shared context. A raw harvested `session.jsonl` bakes in the recording run's session id, cwd, and timestamps; the replay run produces fresh ones. `normalizeSessionLog` scrubs cwd by exact string match, so normalizing the fixture against the *replay* run's cwd would leave the recorded cwd in the header unscrubbed and the compare would fail. The harness therefore derives the fixture's normalize context from its OWN header line (`{ type:'session', id, cwd }`) — `fixtureContext()` in `dsh-acp-snapshot`'s suite module — so both sides scrub to the same `{{sessionId}}`/`{{cwd}}` tokens. An authored fixture copied from the old golden already carries the normalized header (`id:'{{sessionId}}'`, `cwd:'{{cwd}}'`), which yields those tokens as the volatile values and scrubs idempotently. The session-log side uses a plain normalized-string `toEqual`, NOT `toMatchFileSnapshot`, so a run never overwrites the fixture. diff --git a/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md b/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md index 4e4b6a85d7..862dfd42fa 100644 --- a/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md +++ b/docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md @@ -8,7 +8,7 @@ Every model-driving ACP snapshot fixture (`session.jsonl`) embedded the full com ## Decision -Exactly one scenario — `text-turn`, flagged `pinsHeader` in `acp.snapshot.ts` — commits and compares the full request-header content. Every other fixture stores and compares that content as stable tokens via the pure normalizer `scrubRequestHeaders` in `snapshot-normalize.ts`: a `request/header` event's `header.system` becomes `"{{system}}"` and `header.tools` becomes `"{{tools}}"`; a `request/header-delta` keeps its structural facts — the system delta's `keepStart`/`keepEnd` line positions with one `{{system}}` token per inserted line, the tools delta's added/removed/changed tool names — and tokenizes only the bulk (prompt text, schema bodies), so two different deltas still compare different. The scrub is composed in front of `normalizeSessionLog` on BOTH sides of a non-pinning scenario's log compare and applied to the harvested logs record mode writes, so a re-record cannot smuggle the content back. Absent fields stay absent — WHETHER a header carried a prompt or tools is behavior and stays visible — and `config`/`reason` stay verbatim: a model swap churns every fixture by design (it invalidates the recorded responses), while a prompt or schema edit churns none of them (replay derives model behavior exclusively from `assistant/chunk` events and never reads header content — see `dsh-llm-replay`). +Exactly one scenario — `text-turn`, flagged `pinsHeader` in the `acp.snapshot.ts` scenario table — commits and compares the full request-header content; the pin mechanics live in [`dsh-acp-snapshot`](../../../../packages/support/acp-snapshot/README.md), whose suite factory enforces one pin per consuming suite. Every other fixture stores and compares that content as stable tokens via the pure normalizer `scrubRequestHeaders` in that package's `normalize.ts`: a `request/header` event's `header.system` becomes `"{{system}}"` and `header.tools` becomes `"{{tools}}"`; a `request/header-delta` keeps its structural facts — the system delta's `keepStart`/`keepEnd` line positions with one `{{system}}` token per inserted line, the tools delta's added/removed/changed tool names — and tokenizes only the bulk (prompt text, schema bodies), so two different deltas still compare different. The scrub is composed in front of `normalizeSessionLog` on BOTH sides of a non-pinning scenario's log compare and applied to the harvested logs record mode writes, so a re-record cannot smuggle the content back. Absent fields stay absent — WHETHER a header carried a prompt or tools is behavior and stays visible — and `config`/`reason` stay verbatim: a model swap churns every fixture by design (it invalidates the recorded responses), while a prompt or schema edit churns none of them (replay derives model behavior exclusively from `assistant/chunk` events and never reads header content — see `dsh-llm-replay`). A system-prompt or tool-schema change therefore lands as exactly one committed-fixture diff — the pinned `text-turn` header line — updated by hand or by re-recording that one scenario (`pnpm run test:snapshot:record` with `-t text-turn`). diff --git a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md index 2bd061d422..839e1d4542 100644 --- a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md +++ b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md @@ -4,7 +4,7 @@ Status: proposed ## Problem -The ACP snapshot tier ([snapshot RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md)) is built from three modules that live inside one example's test directory: [snapshot-harness.ts](../../../../examples/acp-agent/tests/snapshot-harness.ts) (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), [snapshot-normalize.ts](../../../../examples/acp-agent/tests/snapshot-normalize.ts) (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in [acp.snapshot.ts](../../../../examples/acp-agent/tests/acp.snapshot.ts) (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). +The ACP snapshot tier ([snapshot RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md)) is built from three modules that live inside one example's test directory: `snapshot-harness.ts` (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), `snapshot-normalize.ts` (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in [acp.snapshot.ts](../../../../examples/acp-agent/tests/acp.snapshot.ts) (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). A second ACP example that wants snapshot coverage — the sandbox/approval composition is the immediate consumer — can only copy those modules, forking exactly the logic that must not drift: record write-back, header scrubbing, child-session harvest ordering. The spawn/client glue is already triplicated across [acp.e2e.ts](../../../../examples/acp-agent/tests/acp.e2e.ts), [hooks.e2e.ts](../../../../examples/acp-agent/tests/hooks.e2e.ts), and the harness, marked by `TODO(acp-test-harness)`. diff --git a/docs/testing.md b/docs/testing.md index d7ed14ecb9..a8cf977e85 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -30,4 +30,4 @@ An e2e assertion re-runs the command or re-reads the file externally; a keyword ## When a snapshot test is required -Any change affecting the editor-facing transcript or end-to-end agent UX — the ACP bridge, the loop's observable output, tool presentation — adds or updates a scenario under `examples/acp-agent/tests/snapshots/` (or states in the PR why none applies). New capability seams, lifecycle shapes, or transcript surfaces name their coverage at every tier at plan time and verify the harness can express it — a harness gap is scheduled work, not a mid-build surprise. +Any change affecting the editor-facing transcript or end-to-end agent UX — the ACP bridge, the loop's observable output, tool presentation — adds or updates a scenario in the owning example's snapshot suite (`examples//tests/snapshots/`, a scenario table over the [`dsh-acp-snapshot`](../packages/support/acp-snapshot/README.md) suite factory; `examples/acp-agent` is the primary suite), or states in the PR why none applies. New capability seams, lifecycle shapes, or transcript surfaces name their coverage at every tier at plan time and verify the harness can express it — a harness gap is scheduled work, not a mid-build surprise. diff --git a/examples/acp-agent/tests/acp.e2e.ts b/examples/acp-agent/tests/acp.e2e.ts index 3245fe1750..714791fa3e 100644 --- a/examples/acp-agent/tests/acp.e2e.ts +++ b/examples/acp-agent/tests/acp.e2e.ts @@ -57,8 +57,8 @@ interface Spawned { } // TODO(acp-test-harness): this subprocess/client boot glue is duplicated with -// hooks.e2e.ts and partly with snapshot-harness.ts. Extract one shared ACP test -// launcher before the TSX/env/permission-stub details drift again. +// hooks.e2e.ts and partly with dsh-acp-snapshot's harness. Migrate both e2e +// files onto that launcher before the TSX/env/permission-stub details drift. function spawnAcpAgent(cwd: string, env: NodeJS.ProcessEnv = process.env): Spawned { const child = spawn( process.execPath, diff --git a/examples/acp-agent/tests/acp.snapshot.ts b/examples/acp-agent/tests/acp.snapshot.ts index bde4f4dbb9..b864189a61 100644 --- a/examples/acp-agent/tests/acp.snapshot.ts +++ b/examples/acp-agent/tests/acp.snapshot.ts @@ -1,86 +1,24 @@ -import { readFile, readdir, writeFile } from 'node:fs/promises' -import { existsSync } from 'node:fs' import { fileURLToPath } from 'node:url' import { dirname, join } from 'node:path' -import { describe, expect, it } from 'vitest' -import { type HarvestedLog, type InputScript, runScenario } from './snapshot-harness.ts' -import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from './snapshot-normalize.ts' +import { defineAcpSnapshotSuite, type Scenario } from '@deepseek-ai/dsh-acp-snapshot' /** - * ACP snapshot tests (REPLAY by default, keyless). Each scenario under - * `snapshots//` ships an `input.json` (the client stdin script) and a - * `session.jsonl` fixture; replay boots the real acp-agent subprocess, drives - * it, and diffs the normalized stdout transcript against the committed - * `stdout.golden.jsonl`. For model scenarios it ALSO checks the re-persisted - * session log — against the `session.jsonl` fixture itself, not a separate - * golden: the fixture doubles as the replay source (recorded scenarios) and the - * expected produced log (both sides normalized before comparing). - * - * Request-header content (the composed system prompt + tool schemas riding on - * `request/header` events) is pinned by exactly ONE scenario — the one with - * `pinsHeader` — and scrubbed to `{{system}}`/`{{tools}}` tokens in every - * other fixture and compare, so a prompt or tool-schema edit churns one - * committed line instead of every fixture. A per-run uniformity guard keeps - * the single pin sound: every live header must equal the pinned one, and no - * header-delta may appear outside the pinning scenario (see the - * pinned-header RFC, - * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). - * - * `pnpm run test:snapshot:record` (DSH_SNAPSHOT=record + -u) re-records the - * `session.jsonl` fixtures against the real API and refreshes the stdout golden - * in one pass. + * The acp-agent example's snapshot suite: the scenario table for + * `dsh-acp-snapshot`'s suite factory, which owns every compare/guard mechanic + * (golden + re-persisted-log diffs, record write-back, the pinned-header + * uniformity guard, the fixture guards). Fixtures live under `snapshots//`; + * `pnpm run test:snapshot:record` re-records the `recorded` scenarios against + * the real API. See the package README (packages/support/acp-snapshot) and the + * snapshot RFC, docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. */ -const SNAPSHOTS_DIR = join(dirname(fileURLToPath(import.meta.url)), 'snapshots') -const RECORDING = process.env.DSH_SNAPSHOT === 'record' - -/** A snapshot scenario and how its fixtures are produced. */ -interface Scenario { - name: string - /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ - hasModelTurn: boolean - /** - * Whether the run persists a comparable session log to diff against the - * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn - * always produces a log worth comparing). Set it independently for a scenario - * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked - * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` - * events but never calls the model. - */ - comparesLog?: boolean - /** - * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` - * from the LIVE API. `recorded` scenarios are model-driven and reproducible; - * `authored` scenarios (a hand-written `replay.override.json` sidecar drives - * replay — e.g. a provider error or a cancel, which the live API can't be - * coaxed into deterministically — or a deterministic hook scenario whose - * derived empty script needs no sidecar) are NEVER re-recorded. - */ - recorded: boolean - /** - * How many SUBAGENT child sessions this scenario records beyond the top-level - * one (0 for a single-session scenario). Each child rides in a sibling fixture - * `session..jsonl` (1-based); replay forwards them to `dsh-llm-replay` so - * each child session replays from its own script, and record mode writes the - * harvested child logs back to those files. Defaults to 0. - */ - childSessions?: number - /** - * Whether THIS scenario's fixtures keep the full request-header content (the - * composed system prompt and tool schema list on `request/header` / - * `request/header-delta` events) and compare it verbatim. Exactly one - * scenario pins it; every other scenario stores and compares that content as - * `{{system}}`/`{{tools}}` tokens ({@link scrubRequestHeaders}), so a system - * prompt or tool-schema change shows up as ONE committed-fixture diff, not - * one per scenario. One pin suffices because header composition is - * suite-uniform (parent, spawn child, and fork child all compose the same - * prompt-modulo-cwd and the same tools) — and that premise is ASSERTED, not - * assumed: every non-pinning run's live headers must equal the pinned - * fixture's (normalized), so a session-dependent header (say, a restricted - * subagent toolset) fails loud until it gets its own pinning scenario. - * Defaults to false. - */ - pinsHeader?: boolean +// The dsh-acp-agent bin (the demo:acp entry), this example's cordis.yml, and +// the repo-root tsconfig (four levels up from examples/acp-agent/tests) — all +// ABSOLUTE: the subprocess cwd is a temp dir outside the repo. +const AGENT = { + binScript: fileURLToPath(new URL('../../../packages/ui/acp-agent/src/bin.ts', import.meta.url)), + configPath: fileURLToPath(new URL('../cordis.yml', import.meta.url)), + tsconfigPath: fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)), } const SCENARIOS: Scenario[] = [ @@ -148,238 +86,9 @@ const SCENARIOS: Scenario[] = [ { name: 'hook-codex-stop-continue', hasModelTurn: true, recorded: true }, ] -/** The single header-pinning scenario. Guarded here (and by a meta-test) so the pin cannot silently vanish. */ -const pinningScenario = SCENARIOS.find(s => s.pinsHeader === true) -if (pinningScenario === undefined) throw new Error('acp.snapshot: no scenario pins the request-header content') - -/** The sibling child-fixture paths for a scenario (`session.1.jsonl` …). */ -function childFixturePaths(dir: string, childSessions: number): string[] { - return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) -} - -/** - * Derive the {@link NormalizeContext} for a `session.jsonl` fixture from its own - * header line (`{ type: 'session', id, cwd }`). A committed fixture carries the - * session id and cwd of the run that harvested it — different from the live - * replay run — so normalizing it against the live run's ctx would leave those - * recorded values unscrubbed. Reading them from the header scrubs the fixture's - * own id/cwd to the same `{{sessionId}}`/`{{cwd}}` tokens the replay output gets. - * An authored fixture whose header is already normalized (`id:'{{sessionId}}'`, - * `cwd:'{{cwd}}'`) yields those tokens as the volatile values, so scrubbing them - * is an idempotent no-op. A header with no `cwd` falls back to a sentinel that - * cannot occur in a log (NOT `''`, which `String.split` would match on every - * character boundary and corrupt the output). - */ -function fixtureContext(fixture: string): NormalizeContext { - const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' - const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } - return { - sessionIds: typeof header.id === 'string' ? [header.id] : [], - cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', - } -} - -/** - * The `data.header` payload of every `request/header` event in a session - * JSONL, in log order, with the log's volatile values scrubbed first - * ({@link normalizeSessionLog}) so headers harvested from different runs — - * each embedding its own temp cwd in the composed prompt — compare on equal - * footing. - */ -function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { - return normalizeSessionLog(rawLog, ctx) - .split('\n') - .filter(line => line.trim().length > 0) - .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) - .filter(record => record.type === 'request/header') - .map(record => record.data?.header) -} - -/** Count the `request/header-delta` events in a session JSONL. */ -function headerDeltaCount(rawLog: string): number { - return rawLog.split('\n') - .filter(line => line.trim().length > 0) - .filter(line => (JSON.parse(line) as { type?: unknown }).type === 'request/header-delta') - .length -} - -for (const scenario of SCENARIOS) { - describe(`snapshot: ${scenario.name}`, () => { - // In RECORD mode, only re-run the `recorded` (live-API) scenarios; the - // `authored` ones (sidecar-driven errors/cancel) are never re-recorded. - it.skipIf(RECORDING && !scenario.recorded)('matches the goldens', async () => { - const dir = join(SNAPSHOTS_DIR, scenario.name) - const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript - const overrideFile = join(dir, 'replay.override.json') - const workspaceDir = join(dir, 'workspace') - const childSessions = scenario.childSessions ?? 0 - const result = await runScenario(input, { - mode: RECORDING ? 'record' : 'replay', - fixtureFile: join(dir, 'session.jsonl'), - ...existsSync(overrideFile) ? { overrideFile } : {}, - // In REPLAY, forward the recorded child fixtures so each subagent session - // replays from its own script. In RECORD they are harvested, not read. - ...!RECORDING && childSessions > 0 ? { childFiles: childFixturePaths(dir, childSessions) } : {}, - ...existsSync(workspaceDir) ? { workspaceDir } : {}, - }) - - // Scrub every volatile id the run produced: the ACP server-issued session - // id plus every harvested log's recorded id (a subagent child id never - // surfaces over ACP, but it appears in the child's own log header). The - // normalizer's UUID catch-all covers any we don't enumerate. - const ctx: NormalizeContext = { - sessionIds: [ - ...result.sessionId !== undefined ? [result.sessionId] : [], - ...result.sessionLogs.map(l => l.id), - ], - cwd: result.cwd, - } - - // RECORD mode (recorded model scenarios only): persist the freshly-harvested - // logs back to their fixtures — the primary to session.jsonl, each child to - // session..jsonl in harvest order. `--update` refreshes the Vitest - // goldens but NOT these fixtures, so write them here. A non-pinning - // scenario's fixtures are written header-scrubbed, so a re-record can - // never smuggle the full prompt/schema content back into every fixture. - const scrub = scenario.pinsHeader === true - ? (log: string): string => log - : scrubRequestHeaders - if (RECORDING && scenario.recorded && scenario.hasModelTurn) { - expect(result.sessionLogs.length, 'record produced no session log to harvest').toBeGreaterThan(0) - expect(result.sessionLogs.length, `expected ${childSessions + 1} session logs (parent + children)`) - .toBe(childSessions + 1) - await writeFile(join(dir, 'session.jsonl'), scrub((result.sessionLogs[0] as HarvestedLog).content)) - for (let i = 1; i < result.sessionLogs.length; i++) { - await writeFile(join(dir, `session.${i}.jsonl`), scrub((result.sessionLogs[i] as HarvestedLog).content)) - } - } - - await expect(normalizeStdout(result.rawStdout, ctx)) - .toMatchFileSnapshot(join(dir, 'stdout.golden.jsonl')) - - // A model turn always produces a log worth comparing; a hook scenario can - // produce one without a model turn (a `rejected` turn carrying `hook/*`). - const comparesLog = scenario.comparesLog ?? scenario.hasModelTurn - if (comparesLog) { - // The harvested logs (primary-first) must match their committed fixtures - // 1:1. Each side passes through normalizeSessionLog, scrubbed against ITS - // OWN volatile values — the live run's via `ctx`, the committed fixture's - // via its own header (a committed file cannot share the live run's ids). - // Unless this scenario pins the header, both sides ALSO pass through - // scrubRequestHeaders: the live log carries the real prompt/schemas, the - // fixture carries the `{{system}}`/`{{tools}}` tokens, and the scrub is - // idempotent — so the compare checks the header's presence, position, - // reason, and config, but not its bulk content (pinned once, in the - // `pinsHeader` scenario). - expect(result.sessionLogs.length, 'this scenario must persist a session log').toBe(childSessions + 1) - const fixtureFiles = ['session.jsonl', ...Array.from({ length: childSessions }, (_, i) => `session.${i + 1}.jsonl`)] - for (let i = 0; i < fixtureFiles.length; i++) { - const harvested = scrub((result.sessionLogs[i] as HarvestedLog).content) - const fixture = scrub(await readFile(join(dir, fixtureFiles[i] as string), 'utf8')) - expect(normalizeSessionLog(harvested, ctx), `${fixtureFiles[i]} mismatch`) - .toEqual(normalizeSessionLog(fixture, fixtureContext(fixture))) - } - } - - // Header-uniformity guard: the single pin is sound only while every - // session in the suite composes the SAME header and keeps it for the - // whole run. Assert both halves live. (1) Every request/header the run - // produced (parent, spawn child, fork child, initial or resume) must - // equal the pinned fixture's header after each side is normalized - // against its own volatile values. (2) No request/header-delta may - // appear at all — a mid-run header change diverges from the pin by - // construction, and its content would be invisible under the scrub. If - // either fails, either the header changed (update the pin: re-record or - // hand-edit the pinning scenario's fixture) or composition became - // session-dependent by design (give the divergent shape its own - // pinning scenario). - if (scenario.pinsHeader !== true) { - const pinnedFixture = await readFile(join(SNAPSHOTS_DIR, pinningScenario.name, 'session.jsonl'), 'utf8') - const pinned = normalizedHeaders(pinnedFixture, fixtureContext(pinnedFixture)) - expect(pinned.length, `the pinning fixture (${pinningScenario.name}) must carry exactly one request/header`) - .toBe(1) - for (const log of result.sessionLogs) { - expect(headerDeltaCount(log.content), `session ${log.id}: a request/header-delta in a non-pinning scenario`) - .toBe(0) - const headers = normalizedHeaders(log.content, ctx) - for (const [k, header] of headers.entries()) { - expect(header, `session ${log.id}: request/header #${k + 1} diverged from the pinned (${pinningScenario.name}) header`) - .toEqual(pinned[0]) - } - } - } - }) - }) -} - -describe('snapshot fixtures', () => { - it('every scenario directory is registered (no orphans)', async () => { - // toMatchFileSnapshot does not prune orphaned golden/fixture files, so a - // renamed/removed scenario could leave a stale dir that nothing exercises. - // Fail loud on any snapshots/ not present in SCENARIOS. - const entries = await readdir(SNAPSHOTS_DIR, { withFileTypes: true }) - const onDisk = entries.filter(e => e.isDirectory()).map(e => e.name).sort() - const registered = SCENARIOS.map(s => s.name).sort() - expect(onDisk).toEqual(registered) - }) - - it('every registered scenario has its required fixture files', async () => { - // Every scenario has an input script and an stdout golden. EVERY scenario - // also needs `session.jsonl`: the harness boots `llm-replay` with that path - // as the replay source for ALL scenarios (acp.snapshot.ts passes - // `fixtureFile: /session.jsonl` unconditionally), and `loadReplayScript` - // throws "fixture not found" when it is absent and no override replaces it. - // A no-model scenario ships a header-only `session.jsonl` (it derives to an - // empty script — no model call is made); a model scenario's fixture also - // doubles as the expected-log artifact the run is diffed against. An authored - // (non-`recorded`) model scenario additionally ships a `replay.override.json` - // sidecar for the throw/hang cases a derived script cannot express. - for (const { name, hasModelTurn, recorded, childSessions } of SCENARIOS) { - const dir = join(SNAPSHOTS_DIR, name) - expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true) - expect(existsSync(join(dir, 'stdout.golden.jsonl')), `${name}/stdout.golden.jsonl`).toBe(true) - expect(existsSync(join(dir, 'session.jsonl')), `${name}/session.jsonl`).toBe(true) - if (hasModelTurn && !recorded) { - expect(existsSync(join(dir, 'replay.override.json')), `${name}/replay.override.json`).toBe(true) - } - // A nested-agent scenario ships one child fixture per recorded subagent - // session (`session.1.jsonl` …), the replay source for that child session. - for (const childFixture of childFixturePaths(dir, childSessions ?? 0)) { - expect(existsSync(childFixture), childFixture).toBe(true) - } - } - }) - - it('exactly one scenario pins the request-header content', () => { - // Zero pins would drop the prompt/schema surface from the suite entirely; - // two would split it. The single pin is the design (pinned-header RFC). - expect(SCENARIOS.filter(s => s.pinsHeader === true).map(s => s.name)).toEqual(['text-turn']) - }) - - it('committed fixtures carry request-header content ONLY in the pinning scenario', async () => { - // The whole point of the pin: a system-prompt or tool-schema change must - // churn exactly one committed line. A non-pinning fixture that carries the - // full header (a hand-recorded file, or a header line hand-edited out of - // its canonical JSON form) silently reopens the suite-wide churn, so fail - // loud here: every non-pinning session*.jsonl must be a fixed point of - // scrubRequestHeaders (apply the scrub to fix a violation), and the - // pinning scenario's fixtures must NOT be (their content IS the pin). - for (const scenario of SCENARIOS) { - const dir = join(SNAPSHOTS_DIR, scenario.name) - const files = [ - 'session.jsonl', - ...Array.from({ length: scenario.childSessions ?? 0 }, (_, i) => `session.${i + 1}.jsonl`), - ] - for (const file of files) { - const fixture = await readFile(join(dir, file), 'utf8') - if (scenario.pinsHeader === true) { - expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} must PIN the full header content`) - .not.toEqual(fixture) - } else { - expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} carries unscrubbed header content`) - .toEqual(fixture) - } - } - } - }) +defineAcpSnapshotSuite({ + agent: AGENT, + snapshotsDir: join(dirname(fileURLToPath(import.meta.url)), 'snapshots'), + scenarios: SCENARIOS, + mode: process.env.DSH_SNAPSHOT === 'record' ? 'record' : 'replay', }) diff --git a/knip.json b/knip.json index 89cd2fffe7..b5ff3c2e5d 100644 --- a/knip.json +++ b/knip.json @@ -9,7 +9,7 @@ "examples/echo-agent/tests/**/*.e2e.ts", "examples/coding-agent/tests/**/*.e2e.ts", "examples/acp-agent/tests/**/*.e2e.ts", - "examples/acp-agent/tests/**/*.snapshot.ts" + "examples/*/tests/**/*.snapshot.ts" ], "project": ["scripts/**/*.ts", "examples/**/*.ts"] }, @@ -21,6 +21,11 @@ "project": ["src/**/*.ts"], "ignoreDependencies": ["cordis"] }, + "packages/support/acp-snapshot": { + "entry": ["tests/**/*.spec.ts"], + "project": ["src/**/*.ts", "tests/**/*.ts"], + "ignoreDependencies": ["cordis"] + }, "packages/core/agent-loop": { "entry": ["tests/**/*.spec.ts", "tests/**/*.e2e.ts"], "project": ["src/**/*.ts", "tests/**/*.ts"] diff --git a/packages/support/README.md b/packages/support/README.md index 233a32d77f..2a08063bad 100644 --- a/packages/support/README.md +++ b/packages/support/README.md @@ -4,8 +4,9 @@ Packages that exist to serve development, testing, and the examples rather than | Package | Role | ctx key | |---|---|---| +| `acp-snapshot/` | ACP snapshot suite kit: subprocess scenario harness + golden normalizers + the `defineAcpSnapshotSuite` factory | (library — imported by example `*.snapshot.ts` suites) | | `invariants/` | Dev-mode event-contract invariants + session-log freeze | (listens on `session/*`, `agent/*`) | | `llm-replay/` | Record/replay adapter: short-circuits `llm/stream` from a recorded session JSONL (keyless snapshot tests) | (listens on `llm/stream`) | | `subagent-mock/` | Scripted `SubagentProvider` for deterministic seam/tool tests | (registers on `ctx.subagents`) | -`invariants` runs only in dev mode (contract checks, not runtime behavior). `llm-replay` backs the demos and the snapshot test tier under the per-file coverage gate. `subagent-mock` exercises the real `ctx.subagents` load path without a model or child agent. A package graduates OUT of `support/` into a product group only when it gains documented product consumers. +`invariants` runs only in dev mode (contract checks, not runtime behavior). `llm-replay` backs the demos and the snapshot test tier under the per-file coverage gate. `acp-snapshot` carries the snapshot tier's harness/normalizer/suite machinery so every example's suite is a scenario table over one shared, gate-covered implementation. `subagent-mock` exercises the real `ctx.subagents` load path without a model or child agent. A package graduates OUT of `support/` into a product group only when it gains documented product consumers. diff --git a/packages/support/acp-snapshot/README.md b/packages/support/acp-snapshot/README.md new file mode 100644 index 0000000000..324cfa9f1d --- /dev/null +++ b/packages/support/acp-snapshot/README.md @@ -0,0 +1,36 @@ +# `@deepseek-ai/dsh-acp-snapshot` + +The ACP snapshot suite kit: the shared machinery behind the keyless snapshot tier (`pnpm run test:snapshot`, [testing policy](../../../docs/testing.md)). An example gets a full snapshot suite from a scenario table plus a fixtures directory; every compare/guard mechanic lives here, under the per-file coverage gate, instead of being copied per example. + +Three layers, importable separately: + +- **`runScenario` (harness)** — boots the real agent bin as a subprocess via tsx (unbuilt, Loader path), drives it over ACP JSON-RPC stdio from a deterministic `input.json` script, tees raw stdout for the golden + purity check, and harvests every persisted session JSONL (parent + subagent children, primary-first) after a graceful stdin-EOF shutdown. Parameterized by `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath` — absolute paths; the subprocess cwd is a temp dir outside the repo). +- **Normalizers** — pure functions turning the two captured surfaces into stable text: `normalizeStdout` (JSON-RPC ids → first-seen sequence; UUIDs/cwd → tokens; doubles as the stdout-purity check), `normalizeSessionLog` (times zeroed, `seq` kept), and the composable `scrubRequestHeaders` (header bulk → `{{system}}`/`{{tools}}`, structure kept — [pinned-header RFC](../../../docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)). +- **`defineAcpSnapshotSuite` (factory)** — registers the whole describe/it tree for a scenario table: per-scenario golden + re-persisted-log compares, record-mode fixture write-back, the per-suite header pin with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin, non-pinning fixtures header-scrubbed). Must be called at vitest collection time. + +A consuming `*.snapshot.ts` is the scenario table plus one factory call: + +```ts +import { dirname, join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { defineAcpSnapshotSuite, type Scenario } from '@deepseek-ai/dsh-acp-snapshot' + +const SCENARIOS: Scenario[] = [ + { name: 'text-turn', hasModelTurn: true, recorded: true, pinsHeader: true }, +] + +defineAcpSnapshotSuite({ + agent: { // absolute paths, resolved from the suite's own location + binScript: fileURLToPath(new URL('../../../packages/ui/acp-agent/src/bin.ts', import.meta.url)), + configPath: fileURLToPath(new URL('../cordis.yml', import.meta.url)), + tsconfigPath: fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)), + }, + snapshotsDir: join(dirname(fileURLToPath(import.meta.url)), 'snapshots'), + scenarios: SCENARIOS, // exactly one entry sets pinsHeader + mode: process.env.DSH_SNAPSHOT === 'record' ? 'record' : 'replay', +}) +``` + +The example also ships a `cordis.snapshot.yml` replay overlay next to its `cordis.yml` (the bin swaps them under `DSH_SNAPSHOT=replay` — [single-source replay config RFC](../../../docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md)); replay fixtures are served by [`dsh-llm-replay`](../llm-replay/README.md), which this package points at via the `DSH_SNAPSHOT_*` env vars it sets on the child. Fixture roles, record/replay semantics, and scenario-table fields are documented on `Scenario` and in the [snapshot RFC](../../../docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md). + +Constraints: `suite.ts` imports vitest, so the package is importable only inside a vitest run (the harness and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the harness speaks the SDK's `ClientSideConnection` and answers `requestPermission` with `cancelled`. diff --git a/packages/support/acp-snapshot/package.json b/packages/support/acp-snapshot/package.json new file mode 100644 index 0000000000..363bc86e25 --- /dev/null +++ b/packages/support/acp-snapshot/package.json @@ -0,0 +1,35 @@ +{ + "name": "@deepseek-ai/dsh-acp-snapshot", + "description": "ACP snapshot suite kit: real-subprocess scenario harness, golden normalizers, and the suite factory behind the keyless snapshot tier", + "version": "0.0.1", + "private": true, + "type": "module", + "main": "lib/index.js", + "types": "lib/types/index.d.ts", + "exports": { + ".": { + "types": "./lib/types/index.d.ts", + "default": "./lib/index.js" + }, + "./src/*": "./src/*", + "./package.json": "./package.json" + }, + "files": [ + "lib/index.js", + "lib/types/**/*.d.ts", + "lib/types/**/*.d.ts.map", + "src" + ], + "license": "BSD-3-Clause", + "dependencies": { + "@agentclientprotocol/sdk": "0.25.1", + "tsx": "^4.22.4", + "vitest": "^4.1.8" + }, + "peerDependencies": { + "cordis": "^4.0.0-rc.6" + }, + "devDependencies": { + "cordis": "^4.0.0-rc.6" + } +} diff --git a/examples/acp-agent/tests/snapshot-harness.ts b/packages/support/acp-snapshot/src/harness.ts similarity index 86% rename from examples/acp-agent/tests/snapshot-harness.ts rename to packages/support/acp-snapshot/src/harness.ts index 8285b870bf..a46f752d5c 100644 --- a/examples/acp-agent/tests/snapshot-harness.ts +++ b/packages/support/acp-snapshot/src/harness.ts @@ -1,16 +1,19 @@ /** - * Shared harness for the ACP snapshot tests. A plain module (NOT a *.spec.ts / - * *.snapshot.ts) so importing it never re-registers another file's tests. + * Shared subprocess harness for ACP snapshot suites. A library module driven by + * the suite factory in ./suite.ts (and directly by harness-level specs); each + * example's `*.snapshot.ts` names its own agent-under-test paths. * - * It boots the REAL examples/acp-agent subprocess via the cordis Loader (so the + * It boots the REAL agent bin subprocess via the cordis Loader (so the * export-shape bug class stays guarded — see docs/postmortem/0001), drives it * over real ACP JSON-RPC stdio with a deterministic input script, tees raw * stdout (for the golden + a purity check) into an SDK `ClientSideConnection`, * and — in record mode — harvests the persisted session JSONL after a graceful - * shutdown flush. Two pure normalizers turn the captured stdout frames and the - * session-log events into stable, snapshot-able text. + * shutdown flush. The pure normalizers in ./normalize.ts turn the captured + * stdout frames and the session-log events into stable, snapshot-able text. * * See docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. + * + * @module @deepseek-ai/dsh-acp-snapshot/harness */ import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process' @@ -31,19 +34,36 @@ import { type SessionNotification, } from '@agentclientprotocol/sdk' -// The dsh-acp-agent bin (the demo:acp entry) and this example's cordis.yml. -// The bin resolves its config-path arg from CWD and, under DSH_SNAPSHOT=replay, -// swaps it for the sibling cordis.snapshot.yml. The child's cwd is a temp dir -// OUTSIDE the repo, so pass the example config's ABSOLUTE path. -const binScript = fileURLToPath(new URL('../../../packages/ui/acp-agent/src/bin.ts', import.meta.url)) -const configPath = fileURLToPath(new URL('../cordis.yml', import.meta.url)) +// Resolve tsx's ESM loader to an ABSOLUTE path once: the child runs with its +// cwd in a temp dir OUTSIDE the repo, where a bare `--import tsx` would not +// resolve from node_modules. import.meta.resolve gives this package's tsx +// regardless of the child cwd. const tsxLoader = fileURLToPath(import.meta.resolve('tsx')) -// The repo-root tsconfig: dev/test run UNBUILT and the `@deepseek-ai/dsh-*` -// imports resolve through its `paths` map. The child's cwd is a temp dir -// OUTSIDE the repo, so tsx's upward search would miss it — point tsx at the -// repo tsconfig explicitly (same fix the e2e harness uses). Repo root is four -// levels up from this file (examples/acp-agent/tests). -const repoTsconfig = fileURLToPath(new URL('../../../tsconfig.json', import.meta.url)) + +/** + * The agent composition a scenario runs against: which bin to boot and which + * leaf config it loads. All paths are ABSOLUTE — the subprocess cwd is a temp + * dir outside the repo, so relative resolution would miss; a suite resolves + * them from its own `import.meta.url`. + */ +export interface AgentUnderTest { + /** The agent bin entry (e.g. `packages/ui/acp-agent/src/bin.ts`), run unbuilt via tsx. */ + binScript: string + /** + * The example's live `cordis.yml`. Under `DSH_SNAPSHOT=replay` the bin swaps + * it for the sibling `cordis.snapshot.yml` (the keyless replay overlay), so + * one path serves both modes. + */ + configPath: string + /** + * The repo-root tsconfig whose `paths` map resolves the unbuilt workspace + * imports. Passed to the child as `TSX_TSCONFIG_PATH`: tsx finds a tsconfig + * by searching UP from the child's cwd — a temp dir outside the repo — so + * without the explicit pin the dsh-* imports fail before the bin writes a + * byte. + */ + tsconfigPath: string +} /** * One step of a scenario's deterministic input script (`input.json`). The @@ -57,7 +77,7 @@ const repoTsconfig = fileURLToPath(new URL('../../../tsconfig.json', import.meta * the only way to exercise a cancel deterministically (a plain `prompt` step * awaits the response, which a cancel/hang scenario would block on forever). */ -type InputStep = +export type InputStep = | { op: 'initialize'; terminalOutput?: boolean } | { op: 'newSession' } | { op: 'newSessionExpectError'; additionalDirectories?: string[] } @@ -102,7 +122,10 @@ export interface RunResult { sessionLogs: HarvestedLog[] } -interface RunOptions { +/** How to run one scenario: the agent to boot, the mode, and the fixture wiring. */ +export interface RunOptions { + /** The agent composition to boot. */ + agent: AgentUnderTest /** `replay` (default, keyless) or `record` (real API, harvests the log). */ mode: 'replay' | 'record' /** The recorded session JSONL fixture path (replay reads it; record writes near it). */ @@ -130,6 +153,10 @@ interface RunOptions { * Run a scenario end-to-end against a freshly-spawned subprocess. Owns the * child and its temp dirs; always tears them down. Returns the captured stdout * and (record mode) the harvested session-log path. + * + * @param input The scenario's input script (steps + optional permission answers). + * @param opts The agent to boot, the mode, and the fixture wiring. + * @returns The captured stdout/stderr, session id, temp cwd, and harvested logs. */ export async function runScenario(input: InputScript, opts: RunOptions): Promise { const cwd = await mkdtemp(join(tmpdir(), 'acp-snap-cwd-')) @@ -151,7 +178,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise } const env: NodeJS.ProcessEnv = { ...process.env, - TSX_TSCONFIG_PATH: repoTsconfig, + TSX_TSCONFIG_PATH: opts.agent.tsconfigPath, DSH_SNAPSHOT: opts.mode, DSH_SNAPSHOT_FILE: opts.fixtureFile, DSH_SNAPSHOT_SESSIONS_ROOT: sessionsRoot, @@ -163,7 +190,7 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise child = spawn( process.execPath, - ['--import', tsxLoader, binScript, configPath], + ['--import', tsxLoader, opts.agent.binScript, opts.agent.configPath], { cwd, env, stdio: ['pipe', 'pipe', 'pipe'] }, ) @@ -302,9 +329,9 @@ async function runStep( // its own). To pin frame order deterministically, wait until the client // has OBSERVED the hang's streamed agent_message_chunk before cancelling — // so those update frames always precede the cancelled prompt response in - // the transcript (without this, the late chunk and the response race; see - // the Codex review of commit 5). Then cancel and await the prompt, which - // the bridge settles as `cancelled` once the abort propagates. + // the transcript (without this, the late chunk and the response race). + // Then cancel and await the prompt, which the bridge settles as + // `cancelled` once the abort propagates. const promptDone = client.prompt({ sessionId, prompt: [{ type: 'text', text: step.text }] }) await waitForUpdate(u => u.sessionUpdate === 'agent_message_chunk') await client.cancel({ sessionId }) @@ -335,8 +362,8 @@ function waitForExit(child: ChildProcessWithoutNullStreams): Promise { * * The JSONL backend lays sessions out as `//.jsonl` * (one bucket per cwd), so a parent and its same-cwd in-process child land in - * the SAME bucket — collecting all files across all buckets catches both (the - * old first-match short-circuit silently dropped the child). Returns `[]` if no + * the SAME bucket — collecting all files across all buckets catches both (a + * first-match short-circuit would silently drop the child). Returns `[]` if no * log was produced (a no-session scenario). */ async function harvestSessionLogs(root: string): Promise { diff --git a/packages/support/acp-snapshot/src/index.ts b/packages/support/acp-snapshot/src/index.ts new file mode 100644 index 0000000000..a0d3380086 --- /dev/null +++ b/packages/support/acp-snapshot/src/index.ts @@ -0,0 +1,37 @@ +/** + * ACP snapshot suite kit — the shared machinery behind the keyless snapshot + * tier (`pnpm run test:snapshot`). Three layers, composable per example: + * the subprocess scenario harness ({@link runScenario}), the pure golden + * normalizers ({@link normalizeStdout} / {@link normalizeSessionLog} / + * {@link scrubRequestHeaders}), and the suite factory + * ({@link defineAcpSnapshotSuite}) that registers a scenario table as a full + * describe/it tree. An example's `*.snapshot.ts` supplies only its + * {@link AgentUnderTest} paths, its snapshots directory, and its + * {@link Scenario} table. + * + * NOTE: ./suite.ts imports vitest, so this package is importable only inside a + * vitest run — a support-tier constraint stated in the README. + * + * @module @deepseek-ai/dsh-acp-snapshot + */ + +export { + runScenario, + type AgentUnderTest, + type HarvestedLog, + type InputScript, + type InputStep, + type RunOptions, + type RunResult, +} from './harness.ts' +export { + normalizeSessionLog, + normalizeStdout, + scrubRequestHeaders, + type NormalizeContext, +} from './normalize.ts' +export { + defineAcpSnapshotSuite, + type Scenario, + type SnapshotSuiteOptions, +} from './suite.ts' diff --git a/examples/acp-agent/tests/snapshot-normalize.ts b/packages/support/acp-snapshot/src/normalize.ts similarity index 91% rename from examples/acp-agent/tests/snapshot-normalize.ts rename to packages/support/acp-snapshot/src/normalize.ts index 28c10f102d..2cbe914b42 100644 --- a/examples/acp-agent/tests/snapshot-normalize.ts +++ b/packages/support/acp-snapshot/src/normalize.ts @@ -15,12 +15,15 @@ * A separate, composable normalizer — {@link scrubRequestHeaders} — replaces * the bulky request-header CONTENT (the composed system prompt and the tool * schema list) with `{{system}}`/`{{tools}}` tokens. It is deliberately NOT - * folded into {@link normalizeSessionLog}: the one header-pinning scenario - * compares that content verbatim, every other scenario composes the scrub in - * (the `pinsHeader` flag in acp.snapshot.ts; see the pinned-header RFC, + * folded into {@link normalizeSessionLog}: each suite's one header-pinning + * scenario compares that content verbatim, every other scenario composes the + * scrub in (the `pinsHeader` flag on the scenario table, consumed by the suite + * factory in ./suite.ts; see the pinned-header RFC, * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). * * See docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md. + * + * @module @deepseek-ai/dsh-acp-snapshot/normalize */ const SESSION_ID = '{{sessionId}}' @@ -69,6 +72,10 @@ function scrubValue(value: unknown, ctx: NormalizeContext): unknown { * (1, 2, 3, …) and all volatile strings scrubbed. Throws if any non-empty line * is not valid JSON — that doubles as the stdout-purity check (no logger leaked * onto the protocol). + * + * @param rawStdout The captured stdout bytes, decoded utf8. + * @param ctx The run's volatile values to scrub. + * @returns The normalized NDJSON transcript, one frame per line. */ export function normalizeStdout(rawStdout: string, ctx: NormalizeContext): string { const lines = rawStdout.split('\n').filter(line => line.trim().length > 0) @@ -97,6 +104,10 @@ export function normalizeStdout(rawStdout: string, ctx: NormalizeContext): strin * zeroed/scrubbed, all volatile strings scrubbed, and `seq` is LEFT INTACT * (deterministic by contract). Output is JSONL in the same shape as the input — * one compact record per line. + * + * @param rawLog The raw session `.jsonl` content. + * @param ctx The run's volatile values to scrub. + * @returns The normalized JSONL log, one record per line. */ export function normalizeSessionLog(rawLog: string, ctx: NormalizeContext): string { const lines = rawLog.split('\n').filter(line => line.trim().length > 0) @@ -140,7 +151,10 @@ export function normalizeSessionLog(rawLog: string, ctx: NormalizeContext): stri * Only lines with something to scrub are re-serialized; every other line * passes through byte-for-byte, so the transform is idempotent and applying * it to an already-scrubbed fixture is a no-op — the on-disk-fixtures guard - * in acp.snapshot.ts relies on exactly that. + * in ./suite.ts relies on exactly that. + * + * @param rawLog The raw session `.jsonl` content. + * @returns The JSONL with header content tokenized, other lines byte-identical. */ export function scrubRequestHeaders(rawLog: string): string { const lines = rawLog.split('\n') diff --git a/packages/support/acp-snapshot/src/suite.ts b/packages/support/acp-snapshot/src/suite.ts new file mode 100644 index 0000000000..48d6692258 --- /dev/null +++ b/packages/support/acp-snapshot/src/suite.ts @@ -0,0 +1,355 @@ +/** + * The ACP snapshot suite factory (REPLAY by default, keyless). A suite is a + * scenario table plus a snapshots directory: each scenario under + * `//` ships an `input.json` (the client stdin script) and + * a `session.jsonl` fixture; replay boots the real agent subprocess + * (./harness.ts), drives it, and diffs the normalized stdout transcript + * against the committed `stdout.golden.jsonl`. For model scenarios it ALSO + * checks the re-persisted session log — against the `session.jsonl` fixture + * itself, not a separate golden: the fixture doubles as the replay source + * (recorded scenarios) and the expected produced log (both sides normalized + * before comparing). + * + * Request-header content (the composed system prompt + tool schemas riding on + * `request/header` events) is pinned by exactly ONE scenario per suite — the + * one with `pinsHeader` — and scrubbed to `{{system}}`/`{{tools}}` tokens in + * every other fixture and compare, so a prompt or tool-schema edit churns one + * committed line instead of every fixture. A per-run uniformity guard keeps + * the single pin sound: every live header must equal the pinned one, and no + * header-delta may appear outside the pinning scenario (see the + * pinned-header RFC, + * docs/rfc/implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md). + * + * `pnpm run test:snapshot:record` (DSH_SNAPSHOT=record + -u) re-records the + * `session.jsonl` fixtures against the real API and refreshes the stdout golden + * in one pass; the caller resolves that env into {@link SnapshotSuiteOptions} + * (env reading stays at the suite edge, not in this library). + * + * @module @deepseek-ai/dsh-acp-snapshot/suite + */ + +import { readFile, readdir, writeFile } from 'node:fs/promises' +import { existsSync } from 'node:fs' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import { type AgentUnderTest, type HarvestedLog, type InputScript, runScenario } from './harness.ts' +import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from './normalize.ts' + +/** A snapshot scenario and how its fixtures are produced. */ +export interface Scenario { + name: string + /** Whether the scenario drives at least one model turn (so a JSONL golden applies). */ + hasModelTurn: boolean + /** + * Whether the run persists a comparable session log to diff against the + * `session.jsonl` fixture. Defaults to {@link hasModelTurn} (a model turn + * always produces a log worth comparing). Set it independently for a scenario + * that produces a non-trivial log WITHOUT a model turn — e.g. a prompt blocked + * by a `UserPromptSubmit` hook, which opens a `rejected` turn carrying `hook/*` + * events but never calls the model. + */ + comparesLog?: boolean + /** + * Whether `test:snapshot:record` regenerates this scenario's `session.jsonl` + * from the LIVE API. `recorded` scenarios are model-driven and reproducible; + * `authored` scenarios (a hand-written `replay.override.json` sidecar drives + * replay — e.g. a provider error or a cancel, which the live API can't be + * coaxed into deterministically — or a deterministic hook scenario whose + * derived empty script needs no sidecar) are NEVER re-recorded. + */ + recorded: boolean + /** + * How many SUBAGENT child sessions this scenario records beyond the top-level + * one (0 for a single-session scenario). Each child rides in a sibling fixture + * `session..jsonl` (1-based); replay forwards them to `dsh-llm-replay` so + * each child session replays from its own script, and record mode writes the + * harvested child logs back to those files. Defaults to 0. + */ + childSessions?: number + /** + * Whether THIS scenario's fixtures keep the full request-header content (the + * composed system prompt and tool schema list on `request/header` / + * `request/header-delta` events) and compare it verbatim. Exactly one + * scenario per suite pins it; every other scenario stores and compares that + * content as `{{system}}`/`{{tools}}` tokens ({@link scrubRequestHeaders}), + * so a system prompt or tool-schema change shows up as ONE committed-fixture + * diff, not one per scenario. One pin suffices because header composition is + * suite-uniform (parent, spawn child, and fork child all compose the same + * prompt-modulo-cwd and the same tools) — and that premise is ASSERTED, not + * assumed: every non-pinning run's live headers must equal the pinned + * fixture's (normalized), so a session-dependent header (say, a restricted + * subagent toolset) fails loud until it gets its own pinning scenario. + * Defaults to false. + */ + pinsHeader?: boolean +} + +/** One suite's inputs: the agent to boot, where its fixtures live, and its scenario table. */ +export interface SnapshotSuiteOptions { + /** The agent composition every scenario boots. */ + agent: AgentUnderTest + /** Absolute path of the suite's `snapshots/` directory (one subdir per scenario). */ + snapshotsDir: string + /** The scenario table; exactly one entry must set `pinsHeader`. */ + scenarios: Scenario[] + /** + * `replay` (keyless, the default tier) or `record` (live API; re-records the + * `recorded` scenarios' fixtures and refreshes the vitest goldens under + * `--update`). The caller derives this from `$DSH_SNAPSHOT` — env reading + * stays outside this library. + */ + mode: 'replay' | 'record' +} + +/** The sibling child-fixture paths for a scenario (`session.1.jsonl` …). */ +function childFixturePaths(dir: string, childSessions: number): string[] { + return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) +} + +/** + * Derive the {@link NormalizeContext} for a `session.jsonl` fixture from its own + * header line (`{ type: 'session', id, cwd }`). A committed fixture carries the + * session id and cwd of the run that harvested it — different from the live + * replay run — so normalizing it against the live run's ctx would leave those + * recorded values unscrubbed. Reading them from the header scrubs the fixture's + * own id/cwd to the same `{{sessionId}}`/`{{cwd}}` tokens the replay output gets. + * An authored fixture whose header is already normalized (`id:'{{sessionId}}'`, + * `cwd:'{{cwd}}'`) yields those tokens as the volatile values, so scrubbing them + * is an idempotent no-op. A header with no `cwd` falls back to a sentinel that + * cannot occur in a log (NOT `''`, which `String.split` would match on every + * character boundary and corrupt the output). + */ +function fixtureContext(fixture: string): NormalizeContext { + const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' + const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } + return { + sessionIds: typeof header.id === 'string' ? [header.id] : [], + cwd: typeof header.cwd === 'string' ? header.cwd : '\0no-cwd\0', + } +} + +/** + * The `data.header` payload of every `request/header` event in a session + * JSONL, in log order, with the log's volatile values scrubbed first + * ({@link normalizeSessionLog}) so headers harvested from different runs — + * each embedding its own temp cwd in the composed prompt — compare on equal + * footing. + */ +function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { + return normalizeSessionLog(rawLog, ctx) + .split('\n') + .filter(line => line.trim().length > 0) + .map(line => JSON.parse(line) as { type?: unknown; data?: { header?: unknown } }) + .filter(record => record.type === 'request/header') + .map(record => record.data?.header) +} + +/** Count the `request/header-delta` events in a session JSONL. */ +function headerDeltaCount(rawLog: string): number { + return rawLog.split('\n') + .filter(line => line.trim().length > 0) + .filter(line => (JSON.parse(line) as { type?: unknown }).type === 'request/header-delta') + .length +} + +/** + * Register the suite: one `describe` per scenario (the golden/log compares and + * the header-uniformity guard) plus the fixture guard block (no orphan + * scenario dirs, required files present, exactly one pin, non-pinning fixtures + * header-scrubbed). Must run at vitest collection time — it calls + * `describe`/`it`. Throws immediately if no scenario pins the header (the + * uniformity guard would have nothing to compare against). + * + * @param options The agent, snapshots directory, scenario table, and mode. + */ +export function defineAcpSnapshotSuite(options: SnapshotSuiteOptions): void { + const { agent, snapshotsDir, scenarios, mode } = options + const RECORDING = mode === 'record' + + /** The suite's single header-pinning scenario. Guarded here (and by a meta-test) so the pin cannot silently vanish. */ + const pinningScenario = scenarios.find(s => s.pinsHeader === true) + if (pinningScenario === undefined) throw new Error('acp-snapshot: no scenario pins the request-header content') + + for (const scenario of scenarios) { + describe(`snapshot: ${scenario.name}`, () => { + // In RECORD mode, only re-run the `recorded` (live-API) scenarios; the + // `authored` ones (sidecar-driven errors/cancel) are never re-recorded. + it.skipIf(RECORDING && !scenario.recorded)('matches the goldens', async () => { + const dir = join(snapshotsDir, scenario.name) + const input = JSON.parse(await readFile(join(dir, 'input.json'), 'utf8')) as InputScript + const overrideFile = join(dir, 'replay.override.json') + const workspaceDir = join(dir, 'workspace') + const childSessions = scenario.childSessions ?? 0 + const result = await runScenario(input, { + agent, + mode, + fixtureFile: join(dir, 'session.jsonl'), + ...existsSync(overrideFile) ? { overrideFile } : {}, + // In REPLAY, forward the recorded child fixtures so each subagent session + // replays from its own script. In RECORD they are harvested, not read. + ...!RECORDING && childSessions > 0 ? { childFiles: childFixturePaths(dir, childSessions) } : {}, + ...existsSync(workspaceDir) ? { workspaceDir } : {}, + }) + + // Scrub every volatile id the run produced: the ACP server-issued session + // id plus every harvested log's recorded id (a subagent child id never + // surfaces over ACP, but it appears in the child's own log header). The + // normalizer's UUID catch-all covers any we don't enumerate. + const ctx: NormalizeContext = { + sessionIds: [ + ...result.sessionId !== undefined ? [result.sessionId] : [], + ...result.sessionLogs.map(l => l.id), + ], + cwd: result.cwd, + } + + // RECORD mode (recorded model scenarios only): persist the freshly-harvested + // logs back to their fixtures — the primary to session.jsonl, each child to + // session..jsonl in harvest order. `--update` refreshes the Vitest + // goldens but NOT these fixtures, so write them here. A non-pinning + // scenario's fixtures are written header-scrubbed, so a re-record can + // never smuggle the full prompt/schema content back into every fixture. + const scrub = scenario.pinsHeader === true + ? (log: string): string => log + : scrubRequestHeaders + if (RECORDING && scenario.recorded && scenario.hasModelTurn) { + expect(result.sessionLogs.length, 'record produced no session log to harvest').toBeGreaterThan(0) + expect(result.sessionLogs.length, `expected ${childSessions + 1} session logs (parent + children)`) + .toBe(childSessions + 1) + await writeFile(join(dir, 'session.jsonl'), scrub((result.sessionLogs[0] as HarvestedLog).content)) + for (let i = 1; i < result.sessionLogs.length; i++) { + await writeFile(join(dir, `session.${i}.jsonl`), scrub((result.sessionLogs[i] as HarvestedLog).content)) + } + } + + await expect(normalizeStdout(result.rawStdout, ctx)) + .toMatchFileSnapshot(join(dir, 'stdout.golden.jsonl')) + + // A model turn always produces a log worth comparing; a hook scenario can + // produce one without a model turn (a `rejected` turn carrying `hook/*`). + const comparesLog = scenario.comparesLog ?? scenario.hasModelTurn + if (comparesLog) { + // The harvested logs (primary-first) must match their committed fixtures + // 1:1. Each side passes through normalizeSessionLog, scrubbed against ITS + // OWN volatile values — the live run's via `ctx`, the committed fixture's + // via its own header (a committed file cannot share the live run's ids). + // Unless this scenario pins the header, both sides ALSO pass through + // scrubRequestHeaders: the live log carries the real prompt/schemas, the + // fixture carries the `{{system}}`/`{{tools}}` tokens, and the scrub is + // idempotent — so the compare checks the header's presence, position, + // reason, and config, but not its bulk content (pinned once, in the + // `pinsHeader` scenario). + expect(result.sessionLogs.length, 'this scenario must persist a session log').toBe(childSessions + 1) + const fixtureFiles = ['session.jsonl', ...Array.from({ length: childSessions }, (_, i) => `session.${i + 1}.jsonl`)] + for (let i = 0; i < fixtureFiles.length; i++) { + const harvested = scrub((result.sessionLogs[i] as HarvestedLog).content) + const fixture = scrub(await readFile(join(dir, fixtureFiles[i] as string), 'utf8')) + expect(normalizeSessionLog(harvested, ctx), `${fixtureFiles[i]} mismatch`) + .toEqual(normalizeSessionLog(fixture, fixtureContext(fixture))) + } + } + + // Header-uniformity guard: the single pin is sound only while every + // session in the suite composes the SAME header and keeps it for the + // whole run. Assert both halves live. (1) Every request/header the run + // produced (parent, spawn child, fork child, initial or resume) must + // equal the pinned fixture's header after each side is normalized + // against its own volatile values. (2) No request/header-delta may + // appear at all — a mid-run header change diverges from the pin by + // construction, and its content would be invisible under the scrub. If + // either fails, either the header changed (update the pin: re-record or + // hand-edit the pinning scenario's fixture) or composition became + // session-dependent by design (give the divergent shape its own + // pinning scenario). + if (scenario.pinsHeader !== true) { + const pinnedFixture = await readFile(join(snapshotsDir, pinningScenario.name, 'session.jsonl'), 'utf8') + const pinned = normalizedHeaders(pinnedFixture, fixtureContext(pinnedFixture)) + expect(pinned.length, `the pinning fixture (${pinningScenario.name}) must carry exactly one request/header`) + .toBe(1) + for (const log of result.sessionLogs) { + expect(headerDeltaCount(log.content), `session ${log.id}: a request/header-delta in a non-pinning scenario`) + .toBe(0) + const headers = normalizedHeaders(log.content, ctx) + for (const [k, header] of headers.entries()) { + expect(header, `session ${log.id}: request/header #${k + 1} diverged from the pinned (${pinningScenario.name}) header`) + .toEqual(pinned[0]) + } + } + } + }) + }) + } + + describe('snapshot fixtures', () => { + it('every scenario directory is registered (no orphans)', async () => { + // toMatchFileSnapshot does not prune orphaned golden/fixture files, so a + // renamed/removed scenario could leave a stale dir that nothing exercises. + // Fail loud on any snapshots/ not present in the scenario table. + const entries = await readdir(snapshotsDir, { withFileTypes: true }) + const onDisk = entries.filter(e => e.isDirectory()).map(e => e.name).sort() + const registered = scenarios.map(s => s.name).sort() + expect(onDisk).toEqual(registered) + }) + + it('every registered scenario has its required fixture files', () => { + // Every scenario has an input script and an stdout golden. EVERY scenario + // also needs `session.jsonl`: the suite boots `llm-replay` with that path + // as the replay source for ALL scenarios (the factory passes + // `fixtureFile: /session.jsonl` unconditionally), and `loadReplayScript` + // throws "fixture not found" when it is absent and no override replaces it. + // A no-model scenario ships a header-only `session.jsonl` (it derives to an + // empty script — no model call is made); a model scenario's fixture also + // doubles as the expected-log artifact the run is diffed against. An authored + // (non-`recorded`) model scenario additionally ships a `replay.override.json` + // sidecar for the throw/hang cases a derived script cannot express. + for (const { name, hasModelTurn, recorded, childSessions } of scenarios) { + const dir = join(snapshotsDir, name) + expect(existsSync(join(dir, 'input.json')), `${name}/input.json`).toBe(true) + expect(existsSync(join(dir, 'stdout.golden.jsonl')), `${name}/stdout.golden.jsonl`).toBe(true) + expect(existsSync(join(dir, 'session.jsonl')), `${name}/session.jsonl`).toBe(true) + if (hasModelTurn && !recorded) { + expect(existsSync(join(dir, 'replay.override.json')), `${name}/replay.override.json`).toBe(true) + } + // A nested-agent scenario ships one child fixture per recorded subagent + // session (`session.1.jsonl` …), the replay source for that child session. + for (const childFixture of childFixturePaths(dir, childSessions ?? 0)) { + expect(existsSync(childFixture), childFixture).toBe(true) + } + } + }) + + it('exactly one scenario pins the request-header content', () => { + // Zero pins would drop the prompt/schema surface from the suite entirely; + // two would split it. One pin per suite is the design (pinned-header RFC); + // WHICH scenario pins is the scenario table's reviewable choice. + expect(scenarios.filter(s => s.pinsHeader === true).map(s => s.name)).toEqual([pinningScenario.name]) + }) + + it('committed fixtures carry request-header content ONLY in the pinning scenario', async () => { + // The whole point of the pin: a system-prompt or tool-schema change must + // churn exactly one committed line. A non-pinning fixture that carries the + // full header (a hand-recorded file, or a header line hand-edited out of + // its canonical JSON form) silently reopens the suite-wide churn, so fail + // loud here: every non-pinning session*.jsonl must be a fixed point of + // scrubRequestHeaders (apply the scrub to fix a violation), and the + // pinning scenario's fixtures must NOT be (their content IS the pin). + for (const scenario of scenarios) { + const dir = join(snapshotsDir, scenario.name) + const files = [ + 'session.jsonl', + ...Array.from({ length: scenario.childSessions ?? 0 }, (_, i) => `session.${i + 1}.jsonl`), + ] + for (const file of files) { + const fixture = await readFile(join(dir, file), 'utf8') + if (scenario.pinsHeader === true) { + expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} must PIN the full header content`) + .not.toEqual(fixture) + } else { + expect(scrubRequestHeaders(fixture), `${scenario.name}/${file} carries unscrubbed header content`) + .toEqual(fixture) + } + } + } + }) + }) +} diff --git a/examples/acp-agent/tests/snapshot-normalize.spec.ts b/packages/support/acp-snapshot/tests/normalize.spec.ts similarity index 98% rename from examples/acp-agent/tests/snapshot-normalize.spec.ts rename to packages/support/acp-snapshot/tests/normalize.spec.ts index fa225bd659..56c15306a3 100644 --- a/examples/acp-agent/tests/snapshot-normalize.spec.ts +++ b/packages/support/acp-snapshot/tests/normalize.spec.ts @@ -1,9 +1,9 @@ import { describe, expect, it } from 'vitest' -import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from '../tests/snapshot-normalize.ts' +import { type NormalizeContext, normalizeSessionLog, normalizeStdout, scrubRequestHeaders } from '../src/normalize.ts' /** * Unit tests for the pure snapshot normalizers. Live as a *.spec.ts (runs in - * the default unit gate) and import the harness-side normalizers directly. + * the default unit gate) and import the normalizers directly. */ const ctx: NormalizeContext = { diff --git a/packages/support/acp-snapshot/tsconfig.json b/packages/support/acp-snapshot/tsconfig.json new file mode 100644 index 0000000000..749cb0208e --- /dev/null +++ b/packages/support/acp-snapshot/tsconfig.json @@ -0,0 +1,11 @@ +{ + "extends": "../../../tsconfig.base.json", + "compilerOptions": { + "rootDir": "src", + "outDir": "lib/types" + }, + "include": [ + "src" + ], + "references": [] +} diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index b501bc21ee..9856e99e95 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -748,6 +748,22 @@ importers: specifier: ^4.0.0-rc.6 version: 4.0.0-rc.6(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.4) + packages/support/acp-snapshot: + dependencies: + '@agentclientprotocol/sdk': + specifier: 0.25.1 + version: 0.25.1(zod@4.4.3) + tsx: + specifier: ^4.22.4 + version: 4.22.4 + vitest: + specifier: ^4.1.8 + version: 4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + devDependencies: + cordis: + specifier: ^4.0.0-rc.6 + version: 4.0.0-rc.6(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.4) + packages/support/invariants: devDependencies: '@deepseek-ai/dsh-agent': @@ -5241,6 +5257,14 @@ snapshots: optionalDependencies: vite: 8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + '@vitest/mocker@4.1.8(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0))': + dependencies: + '@vitest/spy': 4.1.8 + estree-walker: 3.0.3 + magic-string: 0.30.21 + optionalDependencies: + vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + '@vitest/pretty-format@4.1.8': dependencies: tinyrainbow: 3.1.0 @@ -6891,6 +6915,21 @@ snapshots: tsx: 4.22.4 yaml: 2.9.0 + vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0): + dependencies: + lightningcss: 1.32.0 + picomatch: 4.0.4 + postcss: 8.5.15 + rolldown: 1.0.3 + tinyglobby: 0.2.17 + optionalDependencies: + '@types/node': 25.9.3 + esbuild: 0.28.1 + fsevents: 2.3.3 + jiti: 2.7.0 + tsx: 4.22.4 + yaml: 2.9.0 + vitest@4.1.8(@types/node@22.20.0)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@22.20.0)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): dependencies: '@vitest/expect': 4.1.8 @@ -6920,6 +6959,35 @@ snapshots: transitivePeerDependencies: - msw + vitest@4.1.8(@types/node@25.9.3)(@vitest/coverage-v8@4.1.8)(jsdom@29.1.1)(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)): + dependencies: + '@vitest/expect': 4.1.8 + '@vitest/mocker': 4.1.8(vite@8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0)) + '@vitest/pretty-format': 4.1.8 + '@vitest/runner': 4.1.8 + '@vitest/snapshot': 4.1.8 + '@vitest/spy': 4.1.8 + '@vitest/utils': 4.1.8 + es-module-lexer: 2.1.0 + expect-type: 1.3.0 + magic-string: 0.30.21 + obug: 2.1.3 + pathe: 2.0.3 + picomatch: 4.0.4 + std-env: 4.1.0 + tinybench: 2.9.0 + tinyexec: 1.2.4 + tinyglobby: 0.2.17 + tinyrainbow: 3.1.0 + vite: 8.0.16(@types/node@25.9.3)(esbuild@0.28.1)(jiti@2.7.0)(tsx@4.22.4)(yaml@2.9.0) + why-is-node-running: 2.3.0 + optionalDependencies: + '@types/node': 25.9.3 + '@vitest/coverage-v8': 4.1.8(vitest@4.1.8) + jsdom: 29.1.1 + transitivePeerDependencies: + - msw + w3c-xmlserializer@5.0.0: dependencies: xml-name-validator: 5.0.0 diff --git a/tsconfig.build.json b/tsconfig.build.json index b6ba7901f2..96d87c01e9 100644 --- a/tsconfig.build.json +++ b/tsconfig.build.json @@ -44,6 +44,7 @@ { "path": "./packages/ui/app-boot" }, { "path": "./packages/ui/stdio-agent" }, { "path": "./packages/support/llm-replay" }, + { "path": "./packages/support/acp-snapshot" }, { "path": "./packages/subagent/subagent" }, { "path": "./packages/support/subagent-mock" }, { "path": "./packages/subagent/tool-subagent" }, diff --git a/tsconfig.json b/tsconfig.json index 9cd7aa8a6d..ff737baf04 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -55,6 +55,7 @@ { "path": "./packages/ui/app-boot" }, { "path": "./packages/ui/stdio-agent" }, { "path": "./packages/support/llm-replay" }, + { "path": "./packages/support/acp-snapshot" }, { "path": "./packages/subagent/subagent" }, { "path": "./packages/support/subagent-mock" }, { "path": "./packages/subagent/tool-subagent" }, From 610c8e37093d7cc454804c51a8fb58efee66b8cd Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 02:05:06 +0800 Subject: [PATCH 21/24] test(acp-snapshot): fake ACP bin + unit specs to per-file 100% coverage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A scripted fake ACP agent bin (tests/fixtures/fake-acp-agent.ts) speaks real newline JSON-RPC through the REAL runScenario spawn path (tsx loader, temp cwd, env plumbing); every behavior — prompt outcome, session/new rejection, persisted logs, filesystem noise — comes from a behavior.json beside the fixture, so specs script whole subprocess runs from data. harness.spec.ts drives every step op, both expect-error arms, the permission-stub default, env forwarding, workspace seeding, and the harvest ordering/noise/fallback branches. suite.spec.ts runs the factory for real at collection time: a replay suite over committed synthetic fixtures and a record suite over a temp copy (write-back never touches the committed tree; ACP_SNAPSHOT_SPEC_BOOTSTRAP=1 re-bootstraps it), plus direct cases for the exported pure helpers. The suite factory's pure helpers (childFixturePaths, fixtureContext, normalizedHeaders, headerDeltaCount) are exported for those direct specs. Two branches carry justified v8 ignores, both structurally unreachable: the waiter in-bounds guard (noUncheckedIndexedAccess) and waitForExit's already-exited race guard (both call sites sit one synchronous frame after stdin.end()/kill()). The fake bin substitutes the session/new cwd, not process.cwd(), into scripted logs — the realpath difference (/private on darwin) is exactly what the real bin's header carries. packages/support/acp-snapshot/src is at 100% statements, branches, functions, and lines under the per-file gate. --- knip.json | 2 +- packages/support/acp-snapshot/src/harness.ts | 11 +- packages/support/acp-snapshot/src/suite.ts | 30 ++- .../tests/fixtures/fake-acp-agent.ts | 232 +++++++++++++++++ .../record-suite/rec-child/behavior.json | 13 + .../record-suite/rec-child/input.json | 1 + .../record-suite/rec-child/session.1.jsonl | 2 + .../record-suite/rec-child/session.jsonl | 2 + .../rec-child/stdout.golden.jsonl | 4 + .../record-suite/rec-pin/behavior.json | 10 + .../fixtures/record-suite/rec-pin/input.json | 1 + .../record-suite/rec-pin/session.jsonl | 2 + .../record-suite/rec-pin/stdout.golden.jsonl | 4 + .../record-suite/rec-skip/behavior.json | 1 + .../fixtures/record-suite/rec-skip/input.json | 1 + .../rec-skip/replay.override.json | 1 + .../record-suite/rec-skip/session.jsonl | 1 + .../record-suite/rec-skip/stdout.golden.jsonl | 1 + .../suite/authored-error/behavior.json | 10 + .../fixtures/suite/authored-error/input.json | 1 + .../suite/authored-error/replay.override.json | 1 + .../suite/authored-error/session.jsonl | 2 + .../suite/authored-error/stdout.golden.jsonl | 4 + .../fixtures/suite/blocked-log/behavior.json | 10 + .../fixtures/suite/blocked-log/input.json | 1 + .../fixtures/suite/blocked-log/session.jsonl | 2 + .../suite/blocked-log/stdout.golden.jsonl | 4 + .../fixtures/suite/no-model/behavior.json | 1 + .../tests/fixtures/suite/no-model/input.json | 1 + .../fixtures/suite/no-model/session.jsonl | 1 + .../suite/no-model/stdout.golden.jsonl | 1 + .../fixtures/suite/pin-turn/behavior.json | 11 + .../tests/fixtures/suite/pin-turn/input.json | 1 + .../fixtures/suite/pin-turn/session.jsonl | 3 + .../suite/pin-turn/stdout.golden.jsonl | 4 + .../fixtures/suite/plain-turn/behavior.json | 15 ++ .../fixtures/suite/plain-turn/input.json | 1 + .../fixtures/suite/plain-turn/session.1.jsonl | 2 + .../fixtures/suite/plain-turn/session.jsonl | 3 + .../suite/plain-turn/stdout.golden.jsonl | 5 + .../suite/plain-turn/workspace/seed.txt | 1 + .../acp-snapshot/tests/harness.spec.ts | 233 ++++++++++++++++++ .../acp-snapshot/tests/normalize.spec.ts | 45 ++++ .../support/acp-snapshot/tests/suite.spec.ts | 145 +++++++++++ 44 files changed, 819 insertions(+), 8 deletions(-) create mode 100644 packages/support/acp-snapshot/tests/fixtures/fake-acp-agent.ts create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.1.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/replay.override.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/authored-error/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/authored-error/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/authored-error/replay.override.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/authored-error/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/no-model/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/no-model/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/no-model/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/behavior.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/input.json create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.1.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.golden.jsonl create mode 100644 packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/workspace/seed.txt create mode 100644 packages/support/acp-snapshot/tests/harness.spec.ts create mode 100644 packages/support/acp-snapshot/tests/suite.spec.ts diff --git a/knip.json b/knip.json index b5ff3c2e5d..8c0f71f3af 100644 --- a/knip.json +++ b/knip.json @@ -22,7 +22,7 @@ "ignoreDependencies": ["cordis"] }, "packages/support/acp-snapshot": { - "entry": ["tests/**/*.spec.ts"], + "entry": ["tests/**/*.spec.ts", "tests/fixtures/fake-acp-agent.ts"], "project": ["src/**/*.ts", "tests/**/*.ts"], "ignoreDependencies": ["cordis"] }, diff --git a/packages/support/acp-snapshot/src/harness.ts b/packages/support/acp-snapshot/src/harness.ts index a46f752d5c..652e60a469 100644 --- a/packages/support/acp-snapshot/src/harness.ts +++ b/packages/support/acp-snapshot/src/harness.ts @@ -224,7 +224,12 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise sessionUpdate(params: SessionNotification): Promise { for (let i = updateWaiters.length - 1; i >= 0; i--) { const waiter = updateWaiters[i] - if (waiter !== undefined && waiter.match(params.update)) { + // The index is always in-bounds (i only decreases; splice removes at + // i, so lower entries stay valid); the guard satisfies + // noUncheckedIndexedAccess. + /* v8 ignore next 1 -- unreachable in-bounds guard, see above */ + if (waiter === undefined) continue + if (waiter.match(params.update)) { updateWaiters.splice(i, 1) waiter.resolve() } @@ -351,6 +356,10 @@ async function runStep( /** Resolve once the child process exits (any code/signal). */ function waitForExit(child: ChildProcessWithoutNullStreams): Promise { + // Race guard: both call sites run within one synchronous frame of + // stdin.end()/kill(), so the exit event cannot have been delivered yet; + // kept for any future caller that awaits in between. + /* v8 ignore next 1 -- unreachable race guard, see above */ if (child.exitCode !== null || child.signalCode !== null) return Promise.resolve() return new Promise(resolve => child.once('exit', () => { resolve() })) } diff --git a/packages/support/acp-snapshot/src/suite.ts b/packages/support/acp-snapshot/src/suite.ts index 48d6692258..14a4df54da 100644 --- a/packages/support/acp-snapshot/src/suite.ts +++ b/packages/support/acp-snapshot/src/suite.ts @@ -101,8 +101,14 @@ export interface SnapshotSuiteOptions { mode: 'replay' | 'record' } -/** The sibling child-fixture paths for a scenario (`session.1.jsonl` …). */ -function childFixturePaths(dir: string, childSessions: number): string[] { +/** + * The sibling child-fixture paths for a scenario (`session.1.jsonl` …). + * + * @param dir The scenario's snapshots directory (`/`). + * @param childSessions How many subagent child sessions the scenario records. + * @returns One path per child, 1-based, in fixture order. + */ +export function childFixturePaths(dir: string, childSessions: number): string[] { return Array.from({ length: childSessions }, (_, i) => join(dir, `session.${i + 1}.jsonl`)) } @@ -118,8 +124,11 @@ function childFixturePaths(dir: string, childSessions: number): string[] { * is an idempotent no-op. A header with no `cwd` falls back to a sentinel that * cannot occur in a log (NOT `''`, which `String.split` would match on every * character boundary and corrupt the output). + * + * @param fixture The committed `session.jsonl` content. + * @returns The fixture's own volatile values, ready for {@link normalizeSessionLog}. */ -function fixtureContext(fixture: string): NormalizeContext { +export function fixtureContext(fixture: string): NormalizeContext { const firstLine = fixture.split('\n').find(line => line.trim().length > 0) ?? '{}' const header = JSON.parse(firstLine) as { id?: unknown; cwd?: unknown } return { @@ -134,8 +143,12 @@ function fixtureContext(fixture: string): NormalizeContext { * ({@link normalizeSessionLog}) so headers harvested from different runs — * each embedding its own temp cwd in the composed prompt — compare on equal * footing. + * + * @param rawLog The session `.jsonl` content to extract headers from. + * @param ctx The volatile values of the run that produced it. + * @returns The normalized `data.header` payloads, in log order. */ -function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { +export function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { return normalizeSessionLog(rawLog, ctx) .split('\n') .filter(line => line.trim().length > 0) @@ -144,8 +157,13 @@ function normalizedHeaders(rawLog: string, ctx: NormalizeContext): unknown[] { .map(record => record.data?.header) } -/** Count the `request/header-delta` events in a session JSONL. */ -function headerDeltaCount(rawLog: string): number { +/** + * Count the `request/header-delta` events in a session JSONL. + * + * @param rawLog The session `.jsonl` content. + * @returns How many `request/header-delta` events the log carries. + */ +export function headerDeltaCount(rawLog: string): number { return rawLog.split('\n') .filter(line => line.trim().length > 0) .filter(line => (JSON.parse(line) as { type?: unknown }).type === 'request/header-delta') diff --git a/packages/support/acp-snapshot/tests/fixtures/fake-acp-agent.ts b/packages/support/acp-snapshot/tests/fixtures/fake-acp-agent.ts new file mode 100644 index 0000000000..cf41412046 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/fake-acp-agent.ts @@ -0,0 +1,232 @@ +/** + * Scripted fake ACP agent bin for `dsh-acp-snapshot`'s unit specs. Speaks + * newline-delimited JSON-RPC on stdio like the real `dsh-acp-agent` bin, but + * every behavior — how prompts settle, whether session/new rejects, which + * session logs get persisted, what filesystem noise to leave — comes from a + * `behavior.json` sitting NEXT to the `$DSH_SNAPSHOT_FILE` fixture, so a spec + * scripts a whole subprocess run from data. The specs launch it through the + * REAL `runScenario` spawn path (tsx loader, temp cwd, env plumbing), so the + * harness plumbing is exercised for real; only the agent behind the protocol + * is scripted. + * + * The specs (not the golden tier) own this bin: it asserts nothing, echoes + * observable facts into `session/update` text chunks (env probe, permission + * outcome, seeded-workspace listing) for the spec to read off `rawStdout`, and + * exits 0 on stdin EOF after writing the scripted logs — mirroring the real + * bin's dispose-flush-exit shape. + */ + +import { mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs' +import { readdirSync } from 'node:fs' +import { dirname, join } from 'node:path' +import { randomUUID } from 'node:crypto' +import { createInterface } from 'node:readline' + +/** One scripted session log: a file path under the sessions root plus its JSONL lines. */ +interface ScriptedLog { + /** Path relative to `$DSH_SNAPSHOT_SESSIONS_ROOT`, e.g. `bucket/a.jsonl` (an empty dir segment is invalid). */ + file: string + /** + * The JSONL records. String templates `{{CWD}}` and `{{SID}}` are replaced + * with the run's real cwd and the ACP session id this bin issued, so a + * written log carries genuine volatile values for the normalizers to scrub. + */ + lines: unknown[] +} + +/** The whole scripted behavior for one run. Every field defaults to the least surprising choice. */ +interface Behavior { + /** Reject every `session/new` (exercises the expect-error step without extra dirs). */ + rejectNewSession?: boolean + /** Reject `session/new` only when `additionalDirectories` is non-empty (the real bridge's rule). */ + rejectExtraDirs?: boolean + /** How `session/prompt` settles: a clean response, a JSON-RPC error, or a hang until `session/cancel`. */ + prompt?: 'respond' | 'error' | 'hang-until-cancel' + /** Before responding to a prompt, send a `session/request_permission` request and echo its outcome as a chunk. */ + permissionProbe?: boolean + /** Echo the `DSH_SNAPSHOT_*` env the harness set as a chunk (spec-side env-plumbing assertions). */ + echoEnv?: boolean + /** Echo the sorted cwd listing as a chunk (spec-side workspace-seeding assertions). */ + echoWorkspace?: boolean + /** Write a line to stderr on boot (spec-side stderr-capture assertions). */ + stderrNote?: string + /** Session logs to persist on stdin EOF. */ + logs?: ScriptedLog[] + /** Leave a stray FILE directly under the sessions root (harvest must skip it). */ + strayRootFile?: boolean + /** Leave a stray non-`.jsonl` file inside a bucket (harvest must skip it). */ + strayBucketFile?: boolean + /** Delete the sessions root entirely (harvest must yield no logs). */ + deleteSessionsRoot?: boolean +} + +const sessionsRoot = process.env.DSH_SNAPSHOT_SESSIONS_ROOT ?? '' +const fixtureFile = process.env.DSH_SNAPSHOT_FILE ?? '' +const behavior: Behavior = fixtureFile === '' + ? {} + : JSON.parse(readFileSync(join(dirname(fixtureFile), 'behavior.json'), 'utf8')) as Behavior + +if (behavior.stderrNote !== undefined) process.stderr.write(`${behavior.stderrNote}\n`) + +let nextOutboundId = 1000 +let sessionId = '' +/** + * The cwd the client passed to `session/new` — used verbatim for `{{CWD}}` + * substitution, mirroring the real bin (whose persisted header carries the + * session cwd as given, NOT `process.cwd()`, which the OS realpaths — on + * macOS `/var/folders/…` vs `/private/var/folders/…`). + */ +let sessionCwd = '' +/** The parked prompt request id while `hang-until-cancel` waits for the cancel notification. */ +let parkedPromptId: number | string | null = null +/** Resolvers for permission-probe responses, keyed by outbound request id. */ +const pendingPermission = new Map void>() + +function send(frame: Record): void { + process.stdout.write(`${JSON.stringify({ jsonrpc: '2.0', ...frame })}\n`) +} + +function respond(id: number | string, result: unknown): void { + send({ id, result }) +} + +function respondError(id: number | string, message: string): void { + send({ id, error: { code: -32603, message } }) +} + +function chunk(text: string): void { + send({ + method: 'session/update', + params: { sessionId, update: { sessionUpdate: 'agent_message_chunk', content: { type: 'text', text } } }, + }) +} + +/** Substitute the `{{CWD}}`/`{{SID}}` templates through a scripted log record. */ +function instantiate(value: unknown): unknown { + if (typeof value === 'string') return value.split('{{CWD}}').join(sessionCwd).split('{{SID}}').join(sessionId) + if (Array.isArray(value)) return value.map(instantiate) + if (value !== null && typeof value === 'object') { + const out: Record = {} + for (const [k, v] of Object.entries(value)) out[k] = instantiate(v) + return out + } + return value +} + +async function handlePrompt(id: number | string): Promise { + if ((behavior.prompt ?? 'respond') === 'hang-until-cancel') { + // A thought chunk BEFORE any message chunk: a promptAndCancel waiter + // watches for agent_message_chunk, so this exercises its non-matching + // update path while the waiter is armed. + send({ + method: 'session/update', + params: { sessionId, update: { sessionUpdate: 'agent_thought_chunk', content: { type: 'text', text: 'mulling' } } }, + }) + } + chunk('thinking about it') + if (behavior.echoEnv === true) { + chunk(`env:${JSON.stringify({ + mode: process.env.DSH_SNAPSHOT, + override: process.env.DSH_SNAPSHOT_OVERRIDE ?? null, + childFiles: process.env.DSH_SNAPSHOT_CHILD_FILES ?? null, + })}`) + } + if (behavior.echoWorkspace === true) { + chunk(`workspace:${readdirSync(process.cwd()).sort().join(',')}`) + } + if (behavior.permissionProbe === true) { + const requestId = nextOutboundId++ + const outcome = await new Promise((resolve) => { + pendingPermission.set(requestId, resolve) + send({ + id: requestId, + method: 'session/request_permission', + params: { + sessionId, + toolCall: { toolCallId: 'call_fake_1', title: 'fake tool', kind: 'execute', status: 'pending' }, + options: [ + { optionId: 'opt-allow', name: 'Allow once', kind: 'allow_once' }, + { optionId: 'opt-reject', name: 'Reject once', kind: 'reject_once' }, + ], + }, + }) + }) + chunk(`permission:${JSON.stringify(outcome)}`) + } + switch (behavior.prompt ?? 'respond') { + case 'respond': + respond(id, { stopReason: 'end_turn' }) + return + case 'error': + respondError(id, 'model exploded') + return + case 'hang-until-cancel': + parkedPromptId = id + return + } +} + +function handleFrame(frame: Record): void { + const id = frame.id as number | string | undefined + const method = frame.method as string | undefined + const params = (frame.params ?? {}) as Record + // A response to one of OUR outbound requests (the permission probe). + if (method === undefined && id !== undefined && typeof id === 'number' && pendingPermission.has(id)) { + const resolve = pendingPermission.get(id) as (outcome: unknown) => void + pendingPermission.delete(id) + resolve((frame.result as { outcome?: unknown } | undefined)?.outcome ?? null) + return + } + switch (method) { + case 'initialize': + respond(id as number | string, { protocolVersion: 1, agentCapabilities: { loadSession: false } }) + return + case 'session/new': { + const extra = params.additionalDirectories as unknown[] | undefined + if (behavior.rejectNewSession === true || (behavior.rejectExtraDirs === true && extra !== undefined && extra.length > 0)) { + respondError(id as number | string, 'unsupported workspace scope') + return + } + sessionId = randomUUID() + sessionCwd = typeof params.cwd === 'string' ? params.cwd : process.cwd() + respond(id as number | string, { sessionId }) + return + } + case 'session/prompt': + void handlePrompt(id as number | string) + return + case 'session/cancel': + if (parkedPromptId !== null) { + const parked = parkedPromptId + parkedPromptId = null + respond(parked, { stopReason: 'cancelled' }) + } + return + default: + // Unknown method: a notification is ignored; a request gets an error so + // the SDK never waits forever on a frame this fake doesn't model. + if (id !== undefined) respondError(id, `unhandled method ${String(method)}`) + } +} + +function flushLogsAndExit(): void { + for (const log of behavior.logs ?? []) { + const target = join(sessionsRoot, log.file) + mkdirSync(dirname(target), { recursive: true }) + writeFileSync(target, log.lines.map(l => JSON.stringify(instantiate(l))).join('\n') + '\n') + } + if (behavior.strayRootFile === true) writeFileSync(join(sessionsRoot, 'stray.txt'), 'not a bucket\n') + if (behavior.strayBucketFile === true) { + mkdirSync(join(sessionsRoot, 'bucket-noise'), { recursive: true }) + writeFileSync(join(sessionsRoot, 'bucket-noise', 'notes.txt'), 'not a session log\n') + } + if (behavior.deleteSessionsRoot === true) rmSync(sessionsRoot, { recursive: true, force: true }) + process.exit(0) +} + +const rl = createInterface({ input: process.stdin }) +rl.on('line', (line) => { + if (line.trim().length === 0) return + handleFrame(JSON.parse(line) as Record) +}) +rl.on('close', () => { flushLogsAndExit() }) diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/behavior.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/behavior.json new file mode 100644 index 0000000000..d44a3a9698 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/behavior.json @@ -0,0 +1,13 @@ +{ + "prompt": "respond", + "logs": [ + { "file": "b/parent.jsonl", "lines": [ + { "type": "session", "id": "{{SID}}", "createdAt": 700, "cwd": "{{CWD}}" }, + { "type": "request/header", "seq": 0, "time": 3, "data": { "header": { "config": { "model": "fake" }, "system": "SYS PROMPT", "tools": [{ "name": "t1", "description": "D1", "parameters": { "type": "object" } }] }, "reason": "initial" } } + ]}, + { "file": "b/child.jsonl", "lines": [ + { "type": "session", "id": "abababab-cdcd-4efe-8ada-badabadabada", "createdAt": 800, "cwd": "{{CWD}}", "parentSession": "{{SID}}" }, + { "type": "request/header", "seq": 0, "time": 2, "data": { "header": { "config": { "model": "fake" }, "system": "SYS PROMPT", "tools": [{ "name": "t1", "description": "D1", "parameters": { "type": "object" } }] }, "reason": "initial" } } + ]} + ] +} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/input.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/input.json new file mode 100644 index 0000000000..6d3e49b830 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }, { "op": "newSession" }, { "op": "prompt", "text": "rec child" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.1.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.1.jsonl new file mode 100644 index 0000000000..1caf2610b3 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.1.jsonl @@ -0,0 +1,2 @@ +{"type":"session","id":"abababab-cdcd-4efe-8ada-badabadabada","createdAt":800,"cwd":"/var/folders/2g/b32ct0qn1d728l_v6tdkjytr0000gn/T/acp-snap-cwd-KBQJbW","parentSession":"f6fa7fcf-dd9c-4b39-8815-b25ddcebfd88"} +{"type":"request/header","seq":0,"time":2,"data":{"header":{"config":{"model":"fake"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.jsonl new file mode 100644 index 0000000000..a2beac360d --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/session.jsonl @@ -0,0 +1,2 @@ +{"type":"session","id":"f6fa7fcf-dd9c-4b39-8815-b25ddcebfd88","createdAt":700,"cwd":"/var/folders/2g/b32ct0qn1d728l_v6tdkjytr0000gn/T/acp-snap-cwd-KBQJbW"} +{"type":"request/header","seq":0,"time":3,"data":{"header":{"config":{"model":"fake"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.golden.jsonl new file mode 100644 index 0000000000..f173b45b77 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-child/stdout.golden.jsonl @@ -0,0 +1,4 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"thinking about it"}}}} +{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/behavior.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/behavior.json new file mode 100644 index 0000000000..a24e30d80a --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/behavior.json @@ -0,0 +1,10 @@ +{ + "prompt": "respond", + "logs": [{ + "file": "b/main.jsonl", + "lines": [ + { "type": "session", "id": "{{SID}}", "createdAt": 600, "cwd": "{{CWD}}" }, + { "type": "request/header", "seq": 0, "time": 4, "data": { "header": { "config": { "model": "fake" }, "system": "SYS PROMPT", "tools": [{ "name": "t1", "description": "D1", "parameters": { "type": "object" } }] }, "reason": "initial" } } + ] + }] +} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/input.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/input.json new file mode 100644 index 0000000000..9573d20b27 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }, { "op": "newSession" }, { "op": "prompt", "text": "rec pin" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/session.jsonl new file mode 100644 index 0000000000..109a192083 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/session.jsonl @@ -0,0 +1,2 @@ +{"type":"session","id":"ccdc749f-56f3-4267-9750-598b5c60b7b2","createdAt":600,"cwd":"/var/folders/2g/b32ct0qn1d728l_v6tdkjytr0000gn/T/acp-snap-cwd-nOQ4Gy"} +{"type":"request/header","seq":0,"time":4,"data":{"header":{"config":{"model":"fake"},"system":"SYS PROMPT","tools":[{"name":"t1","description":"D1","parameters":{"type":"object"}}]},"reason":"initial"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.golden.jsonl new file mode 100644 index 0000000000..f173b45b77 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-pin/stdout.golden.jsonl @@ -0,0 +1,4 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"thinking about it"}}}} +{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/behavior.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/behavior.json new file mode 100644 index 0000000000..0967ef424b --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/behavior.json @@ -0,0 +1 @@ +{} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/input.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/input.json new file mode 100644 index 0000000000..d1e94c22eb --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/replay.override.json b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/replay.override.json new file mode 100644 index 0000000000..8ed00c0651 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/replay.override.json @@ -0,0 +1 @@ +[{ "kind": "hang" }] diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/session.jsonl new file mode 100644 index 0000000000..104f2a0df2 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/session.jsonl @@ -0,0 +1 @@ +{"type":"session","id":"{{sessionId}}","createdAt":0,"cwd":"{{cwd}}"} diff --git a/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.golden.jsonl new file mode 100644 index 0000000000..d6a1d2b232 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/record-suite/rec-skip/stdout.golden.jsonl @@ -0,0 +1 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/behavior.json b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/behavior.json new file mode 100644 index 0000000000..808d9672b9 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/behavior.json @@ -0,0 +1,10 @@ +{ + "prompt": "error", + "logs": [{ + "file": "b/main.jsonl", + "lines": [ + { "type": "session", "id": "{{SID}}", "createdAt": 500, "cwd": "{{CWD}}" }, + { "type": "turn/end", "seq": 1, "time": 9, "data": { "error": "model exploded" } } + ] + }] +} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/input.json b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/input.json new file mode 100644 index 0000000000..c281971465 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }, { "op": "newSession" }, { "op": "promptExpectError", "text": "boom" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/replay.override.json b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/replay.override.json new file mode 100644 index 0000000000..e868115f35 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/replay.override.json @@ -0,0 +1 @@ +[{ "kind": "throw", "chunks": [], "message": "model exploded", "code": "PROVIDER" }] diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/session.jsonl new file mode 100644 index 0000000000..36991a214e --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/session.jsonl @@ -0,0 +1,2 @@ +{"type":"session","id":"44444444-3333-4222-8111-000000000000","createdAt":17,"cwd":"/rec/authored-cwd"} +{"type":"turn/end","seq":1,"time":17,"data":{"error":"model exploded"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.golden.jsonl new file mode 100644 index 0000000000..2a1d69bd93 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/authored-error/stdout.golden.jsonl @@ -0,0 +1,4 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"thinking about it"}}}} +{"jsonrpc":"2.0","id":3,"error":{"code":-32603,"message":"model exploded"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/behavior.json b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/behavior.json new file mode 100644 index 0000000000..e0a438297d --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/behavior.json @@ -0,0 +1,10 @@ +{ + "prompt": "error", + "logs": [{ + "file": "b/main.jsonl", + "lines": [ + { "type": "session", "id": "{{SID}}", "createdAt": 400, "cwd": "{{CWD}}" }, + { "type": "hook/result", "seq": 1, "time": 8, "data": { "decision": "block", "durationMs": 37 } } + ] + }] +} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/input.json b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/input.json new file mode 100644 index 0000000000..0a711dca3c --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }, { "op": "newSession" }, { "op": "promptExpectError", "text": "blocked" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/session.jsonl new file mode 100644 index 0000000000..6d8474812d --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/session.jsonl @@ -0,0 +1,2 @@ +{"type":"session","id":"99999999-8888-4777-8666-555555555555","createdAt":13,"cwd":"/rec/blocked-cwd"} +{"type":"hook/result","seq":1,"time":13,"data":{"decision":"block","durationMs":99}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.golden.jsonl new file mode 100644 index 0000000000..2a1d69bd93 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/blocked-log/stdout.golden.jsonl @@ -0,0 +1,4 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"thinking about it"}}}} +{"jsonrpc":"2.0","id":3,"error":{"code":-32603,"message":"model exploded"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/no-model/behavior.json b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/behavior.json new file mode 100644 index 0000000000..0967ef424b --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/behavior.json @@ -0,0 +1 @@ +{} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/no-model/input.json b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/input.json new file mode 100644 index 0000000000..d1e94c22eb --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/no-model/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/session.jsonl new file mode 100644 index 0000000000..104f2a0df2 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/session.jsonl @@ -0,0 +1 @@ +{"type":"session","id":"{{sessionId}}","createdAt":0,"cwd":"{{cwd}}"} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.golden.jsonl new file mode 100644 index 0000000000..d6a1d2b232 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/no-model/stdout.golden.jsonl @@ -0,0 +1 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/behavior.json b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/behavior.json new file mode 100644 index 0000000000..422e0a17e6 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/behavior.json @@ -0,0 +1,11 @@ +{ + "prompt": "respond", + "logs": [{ + "file": "b/main.jsonl", + "lines": [ + { "type": "session", "id": "{{SID}}", "createdAt": 100, "cwd": "{{CWD}}" }, + { "type": "request/header", "seq": 0, "time": 100, "data": { "header": { "config": { "model": "fake" }, "system": "SYS PROMPT", "tools": [{ "name": "t1", "description": "D1", "parameters": { "type": "object" } }] }, "reason": "initial" } }, + { "type": "turn/start", "seq": 1, "time": 100, "data": { "turn": 1 } } + ] + }] +} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/input.json b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/input.json new file mode 100644 index 0000000000..b9e2d9bbc5 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }, { "op": "newSession" }, { "op": "prompt", "text": "pin" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/session.jsonl new file mode 100644 index 0000000000..87bf09c839 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/session.jsonl @@ -0,0 +1,3 @@ +{"type":"session","id":"12121212-3434-4545-8686-787878787878","createdAt":7,"cwd":"/rec/pin-cwd"} +{"type":"request/header","seq":0,"time":7,"data":{"header":{"config":{"model":"fake"},"system":"SYS PROMPT","tools":[{"name":"t1","description":"D1","parameters":{"type":"object"}}]},"reason":"initial"}} +{"type":"turn/start","seq":1,"time":7,"data":{"turn":1}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.golden.jsonl new file mode 100644 index 0000000000..f173b45b77 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/pin-turn/stdout.golden.jsonl @@ -0,0 +1,4 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"thinking about it"}}}} +{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/behavior.json b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/behavior.json new file mode 100644 index 0000000000..d5cbbf9d28 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/behavior.json @@ -0,0 +1,15 @@ +{ + "prompt": "respond", + "echoWorkspace": true, + "logs": [ + { "file": "b/parent.jsonl", "lines": [ + { "type": "session", "id": "{{SID}}", "createdAt": 200, "cwd": "{{CWD}}" }, + { "type": "request/header", "seq": 0, "time": 5, "data": { "header": { "config": { "model": "fake" }, "system": "SYS PROMPT", "tools": [{ "name": "t1", "description": "D1", "parameters": { "type": "object" } }] }, "reason": "initial" } }, + { "type": "assistant/chunk", "seq": 1, "time": 5, "data": { "turn": 1, "step": 1, "chunk": { "type": "text-delta", "index": 0, "text": "hi" } } } + ]}, + { "file": "b/child.jsonl", "lines": [ + { "type": "session", "id": "eeeeeeee-1111-4222-8333-444444444444", "createdAt": 300, "cwd": "{{CWD}}", "parentSession": "{{SID}}" }, + { "type": "request/header", "seq": 0, "time": 6, "data": { "header": { "config": { "model": "fake" }, "system": "SYS PROMPT", "tools": [{ "name": "t1", "description": "D1", "parameters": { "type": "object" } }] }, "reason": "initial" } } + ]} + ] +} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/input.json b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/input.json new file mode 100644 index 0000000000..60b9e363b5 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/input.json @@ -0,0 +1 @@ +{ "steps": [{ "op": "initialize" }, { "op": "newSession" }, { "op": "prompt", "text": "plain" }] } diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.1.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.1.jsonl new file mode 100644 index 0000000000..a844f891fc --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.1.jsonl @@ -0,0 +1,2 @@ +{"type":"session","id":"eeeeeeee-1111-4222-8333-444444444444","createdAt":12,"cwd":"/rec/plain-cwd","parentSession":"56565656-7878-4989-8a9a-9b9b9b9b9b9b"} +{"type":"request/header","seq":0,"time":12,"data":{"header":{"config":{"model":"fake"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.jsonl new file mode 100644 index 0000000000..744998f959 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/session.jsonl @@ -0,0 +1,3 @@ +{"type":"session","id":"56565656-7878-4989-8a9a-9b9b9b9b9b9b","createdAt":11,"cwd":"/rec/plain-cwd"} +{"type":"request/header","seq":0,"time":11,"data":{"header":{"config":{"model":"fake"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}} +{"type":"assistant/chunk","seq":1,"time":11,"data":{"turn":1,"step":1,"chunk":{"type":"text-delta","index":0,"text":"hi"}}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.golden.jsonl b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.golden.jsonl new file mode 100644 index 0000000000..d0242ae39f --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/stdout.golden.jsonl @@ -0,0 +1,5 @@ +{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentCapabilities":{"loadSession":false}}} +{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"thinking about it"}}}} +{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"workspace:seed.txt"}}}} +{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn"}} diff --git a/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/workspace/seed.txt b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/workspace/seed.txt new file mode 100644 index 0000000000..c19e887d68 --- /dev/null +++ b/packages/support/acp-snapshot/tests/fixtures/suite/plain-turn/workspace/seed.txt @@ -0,0 +1 @@ +seeded diff --git a/packages/support/acp-snapshot/tests/harness.spec.ts b/packages/support/acp-snapshot/tests/harness.spec.ts new file mode 100644 index 0000000000..3884e36095 --- /dev/null +++ b/packages/support/acp-snapshot/tests/harness.spec.ts @@ -0,0 +1,233 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { delimiter, join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { afterAll, describe, expect, it } from 'vitest' +import { runScenario, type AgentUnderTest, type InputStep } from '../src/harness.ts' + +/** + * Unit tests for the subprocess harness, driven through the REAL spawn path + * (tsx loader, temp cwd, env plumbing) against the scripted fake ACP bin in + * ./fixtures/fake-acp-agent.ts. Each case writes a `behavior.json` next to a + * throwaway fixture path; the fake bin echoes observable facts (env, seeded + * workspace, permission outcomes) into `agent_message_chunk` text, so the + * assertions read plain `rawStdout`. + */ + +const AGENT: AgentUnderTest = { + binScript: fileURLToPath(new URL('./fixtures/fake-acp-agent.ts', import.meta.url)), + // The fake bin ignores its config argv; any real path documents the shape. + configPath: fileURLToPath(new URL('./fixtures/fake-acp-agent.ts', import.meta.url)), + tsconfigPath: fileURLToPath(new URL('../../../../tsconfig.json', import.meta.url)), +} + +/** Temp scenario dirs to drop after the suite. */ +const tempDirs: string[] = [] +afterAll(async () => { + for (const dir of tempDirs) await rm(dir, { recursive: true, force: true }) +}) + +/** Write a behavior.json into a fresh temp dir; return the sibling fixture path the harness points the bin at. */ +async function scenario(behavior: object): Promise<{ dir: string; fixtureFile: string }> { + const dir = await mkdtemp(join(tmpdir(), 'acp-snap-spec-')) + tempDirs.push(dir) + await writeFile(join(dir, 'behavior.json'), JSON.stringify(behavior)) + return { dir, fixtureFile: join(dir, 'session.jsonl') } +} + +const boot: InputStep[] = [{ op: 'initialize' }, { op: 'newSession' }] + +describe('runScenario', () => { + it('drives a full turn: initialize (terminal caps), session, prompt, permission stub, harvest', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ + permissionProbe: true, + logs: [{ + file: 'bucket/main.jsonl', + lines: [ + { type: 'session', id: '{{SID}}', createdAt: 42, cwd: '{{CWD}}' }, + { type: 'turn/start', seq: 1, time: 9, data: { turn: 1 } }, + ], + }], + }) + const result = await runScenario( + { steps: [{ op: 'initialize', terminalOutput: true }, { op: 'newSession' }, { op: 'prompt', text: 'go' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.sessionId).toBeDefined() + // The harness's client answers a permission request with `cancelled`; the + // fake bin echoes the outcome it received back as a chunk. + expect(result.rawStdout).toContain('permission:{\\"outcome\\":\\"cancelled\\"}') + expect(result.sessionLogs).toHaveLength(1) + expect(result.sessionLogs[0]?.id).toBe(result.sessionId) + expect(result.sessionLogs[0]?.createdAt).toBe(42) + expect(result.sessionLogs[0]?.content).toContain('turn/start') + // The harvested log embeds the run's REAL temp cwd (template-substituted). + expect(result.sessionLogs[0]?.content).toContain(result.cwd) + }) + + it('forwards override/child fixture paths into the child env and captures stderr', { timeout: 20_000 }, async () => { + const { dir, fixtureFile } = await scenario({ echoEnv: true, stderrNote: 'fake bin booted' }) + const childFiles = [join(dir, 'session.1.jsonl'), join(dir, 'session.2.jsonl')] + const result = await runScenario( + { steps: [...boot, { op: 'prompt', text: 'env?' }] }, + { + agent: AGENT, + mode: 'replay', + fixtureFile, + overrideFile: join(dir, 'replay.override.json'), + childFiles, + // A workspaceDir that does not exist is skipped, not an error. + workspaceDir: join(dir, 'no-such-workspace'), + }, + ) + expect(result.stderr).toContain('fake bin booted') + expect(result.rawStdout).toContain('replay.override.json') + // Child paths ride one env var, joined with the platform delimiter. + expect(result.rawStdout).toContain(JSON.stringify(childFiles.join(delimiter)).slice(1, -1)) + }) + + it('seeds the workspace dir into the temp cwd before the run', { timeout: 20_000 }, async () => { + const { dir, fixtureFile } = await scenario({ echoWorkspace: true }) + const workspaceDir = join(dir, 'workspace') + await writeFile(join(dir, 'behavior.json'), JSON.stringify({ echoWorkspace: true })) + const { mkdir } = await import('node:fs/promises') + await mkdir(workspaceDir, { recursive: true }) + await writeFile(join(workspaceDir, 'seeded.txt'), 'hello') + const result = await runScenario( + { steps: [...boot, { op: 'prompt', text: 'ls' }] }, + { agent: AGENT, mode: 'replay', fixtureFile, workspaceDir }, + ) + expect(result.rawStdout).toContain('workspace:seeded.txt') + }) + + it('promptAndCancel waits for the streamed chunk, cancels, and settles the prompt', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ prompt: 'hang-until-cancel' }) + const result = await runScenario( + { steps: [...boot, { op: 'promptAndCancel', text: 'hang' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.rawStdout).toContain('"stopReason":"cancelled"') + // The streamed chunk deterministically precedes the cancelled response. + expect(result.rawStdout.indexOf('thinking about it')).toBeLessThan(result.rawStdout.indexOf('cancelled')) + }) + + it('promptExpectError swallows a model-error response as the expected outcome', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ prompt: 'error' }) + const result = await runScenario( + { steps: [...boot, { op: 'promptExpectError', text: 'boom' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.rawStdout).toContain('model exploded') + }) + + it('promptExpectError throws when the prompt unexpectedly succeeds (and teardown kills the live child)', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ prompt: 'respond' }) + await expect(runScenario( + { steps: [...boot, { op: 'promptExpectError', text: 'fine' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + )).rejects.toThrow(/expected the prompt to fail/) + }) + + it('newSessionExpectError swallows the rejection, with and without extra dirs', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ rejectExtraDirs: true }) + const result = await runScenario( + { steps: [{ op: 'initialize' }, { op: 'newSessionExpectError', additionalDirectories: ['/elsewhere'] }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + // No session was created, so no id and no logs. + expect(result.sessionId).toBeUndefined() + expect(result.sessionLogs).toHaveLength(0) + + const rejectAll = await scenario({ rejectNewSession: true }) + const second = await runScenario( + { steps: [{ op: 'initialize' }, { op: 'newSessionExpectError' }] }, + { agent: AGENT, mode: 'replay', fixtureFile: rejectAll.fixtureFile }, + ) + expect(second.rawStdout).toContain('unsupported workspace scope') + }) + + it('newSessionExpectError throws when session/new unexpectedly succeeds', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({}) + await expect(runScenario( + { steps: [{ op: 'initialize' }, { op: 'newSessionExpectError' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + )).rejects.toThrow(/expected session\/new to be rejected/) + }) + + it('a plain cancel step is forwarded (and ignored by an idle agent)', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({}) + const result = await runScenario( + { steps: [...boot, { op: 'cancel' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.sessionId).toBeDefined() + }) + + it.each([ + [{ op: 'prompt', text: 'x' }, /prompt before newSession/], + [{ op: 'promptExpectError', text: 'x' }, /promptExpectError before newSession/], + [{ op: 'promptAndCancel', text: 'x' }, /promptAndCancel before newSession/], + [{ op: 'cancel' }, /cancel before newSession/], + ] as [InputStep, RegExp][])('rejects %j before newSession', { timeout: 20_000 }, async (step, message) => { + const { fixtureFile } = await scenario({}) + await expect(runScenario( + { steps: [{ op: 'initialize' }, step] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + )).rejects.toThrow(message) + }) + + it('rejects an unknown input op', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({}) + const bogus = { op: 'reticulate' } as unknown as InputStep + await expect(runScenario( + { steps: [bogus] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + )).rejects.toThrow(/unknown input op/) + }) + + it('harvests all logs primary-first, children by createdAt then id, skipping filesystem noise', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ + strayRootFile: true, + strayBucketFile: true, + logs: [ + // File names chosen so readdir feeds the sort children-first AND + // parent-in-the-middle: the comparator then sees a parent on both + // sides of a pair, plus the same-createdAt (localeCompare) tiebreak. + { file: 'b1/aa-child-c.jsonl', lines: [{ type: 'session', id: 'cccccccc-0000-4000-8000-000000000000', createdAt: 500, parentSession: '{{SID}}' }] }, + { file: 'b1/bb-parent.jsonl', lines: [{ type: 'session', id: '{{SID}}', createdAt: 900 }] }, + { file: 'b1/cc-child-a.jsonl', lines: [{ type: 'session', id: 'aaaaaaaa-0000-4000-8000-000000000000', createdAt: 500, parentSession: '{{SID}}' }] }, + // Missing id/createdAt fall back to ''/0; earliest child by createdAt. + { file: 'b2/orphan-fields.jsonl', lines: [{ type: 'session', parentSession: '{{SID}}' }] }, + ], + }) + const result = await runScenario( + { steps: [...boot, { op: 'prompt', text: 'go' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.sessionLogs.map(l => [l.id, l.createdAt])).toEqual([ + [result.sessionId, 900], + ['', 0], + ['aaaaaaaa-0000-4000-8000-000000000000', 500], + ['cccccccc-0000-4000-8000-000000000000', 500], + ]) + expect(result.sessionLogs[1]?.parentSession).toBe(result.sessionId) + }) + + it('treats an empty log file as a header-less primary with default fields', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ logs: [{ file: 'b/empty.jsonl', lines: [] }] }) + const result = await runScenario( + { steps: boot }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.sessionLogs.map(l => [l.id, l.createdAt, l.parentSession])).toEqual([['', 0, undefined]]) + }) + + it('yields no logs when the sessions root vanished', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ deleteSessionsRoot: true }) + const result = await runScenario( + { steps: boot }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.sessionLogs).toHaveLength(0) + }) +}) diff --git a/packages/support/acp-snapshot/tests/normalize.spec.ts b/packages/support/acp-snapshot/tests/normalize.spec.ts index 56c15306a3..8ebd1412b9 100644 --- a/packages/support/acp-snapshot/tests/normalize.spec.ts +++ b/packages/support/acp-snapshot/tests/normalize.spec.ts @@ -107,6 +107,17 @@ describe('normalizeSessionLog', () => { const out = normalizeSessionLog(`${header({})}\n${ev}\n`, ctx) expect(out).toContain('"durationMs":88') }) + + it('tolerates records missing the volatile fields it would zero', () => { + const bareHeader = JSON.stringify({ type: 'session', id: 's' }) + const timeless = JSON.stringify({ type: 'note', seq: 1 }) + const bareHook = JSON.stringify({ type: 'hook/result', seq: 2, time: 5, data: { decision: 'allow' } }) + const nullDataHook = JSON.stringify({ type: 'hook/result', seq: 3, time: 6, data: null }) + const out = normalizeSessionLog(`${bareHeader}\n${timeless}\n${bareHook}\n${nullDataHook}\n`, ctx) + expect(out).toContain('"type":"note","seq":1') + expect(out).toContain('"decision":"allow"') + expect(out).not.toContain('durationMs') + }) }) describe('scrubRequestHeaders', () => { @@ -135,6 +146,40 @@ describe('scrubRequestHeaders', () => { expect(out).not.toContain('{{tools}}') }) + it('scrubs a header carrying only one of system/tools, leaving the other absent', () => { + const systemOnly = scrubRequestHeaders(`${headerLine}\n${headerEvent({ system: 'secret prompt' })}\n`) + expect(systemOnly).toContain('"system":"{{system}}"') + expect(systemOnly).not.toContain('{{tools}}') + const toolsOnly = scrubRequestHeaders(`${headerLine}\n${headerEvent({ tools: [{ name: 't' }] })}\n`) + expect(toolsOnly).toContain('"tools":"{{tools}}"') + expect(toolsOnly).not.toContain('{{system}}') + }) + + it('leaves a delta with no scrubbable payload byte-identical (config-only, or non-array shapes)', () => { + const configOnly = JSON.stringify({ type: 'request/header-delta', seq: 8, time: 9, data: { config: { model: 'm2' } } }) + const oddShapes = JSON.stringify({ type: 'request/header-delta', seq: 9, time: 9, data: { system: { insert: 'not-an-array' }, tools: null } }) + const headerless = JSON.stringify({ type: 'request/header', seq: 10, time: 9, data: { reason: 'initial' } }) + const nullData = JSON.stringify({ type: 'request/header', seq: 11, time: 9, data: null }) + const raw = `${headerLine}\n${configOnly}\n${oddShapes}\n${headerless}\n${nullData}\n` + expect(scrubRequestHeaders(raw)).toBe(raw) + }) + + it('scrubs a one-sided tools delta and passes non-object schema entries through', () => { + const addedOnly = JSON.stringify({ + type: 'request/header-delta', seq: 8, time: 9, + data: { tools: { added: [null, 'weird', { name: 'x', description: 'D' }] } }, + }) + const out = scrubRequestHeaders(`${headerLine}\n${addedOnly}\n`) + // Non-object entries survive untouched; the object entry keeps only name. + expect(out).toContain('"added":[null,"weird",{"name":"x","description":"{{tools}}"}]') + const changedOnly = JSON.stringify({ + type: 'request/header-delta', seq: 8, time: 9, + data: { tools: { changed: [{ name: 'y', parameters: {} }] } }, + }) + expect(scrubRequestHeaders(`${headerLine}\n${changedOnly}\n`)) + .toContain('"changed":[{"name":"y","parameters":"{{tools}}"}]') + }) + it('scrubs a header-delta system payload but keeps its line positions and arity', () => { const delta = JSON.stringify({ type: 'request/header-delta', seq: 8, time: 9, diff --git a/packages/support/acp-snapshot/tests/suite.spec.ts b/packages/support/acp-snapshot/tests/suite.spec.ts new file mode 100644 index 0000000000..0bc15aeec2 --- /dev/null +++ b/packages/support/acp-snapshot/tests/suite.spec.ts @@ -0,0 +1,145 @@ +import { cpSync, mkdtempSync } from 'node:fs' +import { rm } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { afterAll, describe, expect, it } from 'vitest' +import { defineAcpSnapshotSuite, type Scenario } from '../src/index.ts' +import { childFixturePaths, fixtureContext, headerDeltaCount, normalizedHeaders } from '../src/suite.ts' + +/** + * Unit tests for the suite factory, by running it: two synthetic suites over + * the scripted fake ACP bin (./fixtures/fake-acp-agent.ts) register REAL + * describe/it trees at collection time, so every factory path — golden and log + * compares, the per-suite header pin and its uniformity guard, record-mode + * fixture write-back, skip semantics, and the fixture guard block — executes + * as an ordinary green test. The pure helpers get direct cases below. + * + * The replay suite runs against the committed fixtures in ./fixtures/suite. + * The record suite runs against a TEMP COPY of ./fixtures/record-suite + * (record mode writes session fixtures back into its snapshots dir; a run must + * never touch the committed tree). To re-bootstrap the record tree's goldens + * after changing the fake bin's output, run this spec once with + * `ACP_SNAPSHOT_SPEC_BOOTSTRAP=1` (points the record suite at the committed + * tree so vitest creates/updates the goldens and the write-back lands there), + * then commit the result. + */ + +const AGENT = { + binScript: fileURLToPath(new URL('./fixtures/fake-acp-agent.ts', import.meta.url)), + configPath: fileURLToPath(new URL('./fixtures/fake-acp-agent.ts', import.meta.url)), + tsconfigPath: fileURLToPath(new URL('../../../../tsconfig.json', import.meta.url)), +} + +const REPLAY_DIR = fileURLToPath(new URL('./fixtures/suite', import.meta.url)) +const RECORD_SRC = fileURLToPath(new URL('./fixtures/record-suite', import.meta.url)) + +const REPLAY_SCENARIOS: Scenario[] = [ + { name: 'pin-turn', hasModelTurn: true, recorded: true, pinsHeader: true }, + { name: 'plain-turn', hasModelTurn: true, recorded: true, childSessions: 1 }, + { name: 'no-model', hasModelTurn: false, recorded: false }, + { name: 'blocked-log', hasModelTurn: false, comparesLog: true, recorded: false }, + { name: 'authored-error', hasModelTurn: true, recorded: false }, +] + +const RECORD_SCENARIOS: Scenario[] = [ + { name: 'rec-pin', hasModelTurn: true, recorded: true, pinsHeader: true }, + { name: 'rec-child', hasModelTurn: true, recorded: true, childSessions: 1 }, + // recorded:false in record mode → registered but skipped (never re-recorded). + { name: 'rec-skip', hasModelTurn: true, recorded: false }, +] + +// Record mode mutates its snapshots dir, so run it on a throwaway copy — +// except under the documented bootstrap knob, which regenerates the committed +// fixtures/goldens in place. +const BOOTSTRAP = process.env.ACP_SNAPSHOT_SPEC_BOOTSTRAP === '1' +const recordDir = BOOTSTRAP ? RECORD_SRC : mkdtempSync(join(tmpdir(), 'acp-snap-record-suite-')) +if (!BOOTSTRAP) cpSync(RECORD_SRC, recordDir, { recursive: true }) +afterAll(async () => { + if (!BOOTSTRAP) await rm(recordDir, { recursive: true, force: true }) +}) + +describe('defineAcpSnapshotSuite: replay mode', () => { + defineAcpSnapshotSuite({ agent: AGENT, snapshotsDir: REPLAY_DIR, scenarios: REPLAY_SCENARIOS, mode: 'replay' }) +}) + +// The record suite's tests run in registration order: rec-pin re-records the +// pinned fixture FIRST, so rec-child's uniformity guard reads the fresh pin. +describe('defineAcpSnapshotSuite: record mode', () => { + defineAcpSnapshotSuite({ agent: AGENT, snapshotsDir: recordDir, scenarios: RECORD_SCENARIOS, mode: 'record' }) +}) + +describe('defineAcpSnapshotSuite: registration contract', () => { + it('throws when no scenario pins the request-header content', () => { + expect(() => { + defineAcpSnapshotSuite({ + agent: AGENT, + snapshotsDir: REPLAY_DIR, + scenarios: [{ name: 'pinless', hasModelTurn: true, recorded: true }], + mode: 'replay', + }) + }).toThrow(/no scenario pins/) + }) +}) + +describe('childFixturePaths', () => { + it('yields one sibling path per child, 1-based', () => { + expect(childFixturePaths('/snap/s', 2)).toEqual(['/snap/s/session.1.jsonl', '/snap/s/session.2.jsonl']) + }) + + it('yields nothing for a single-session scenario', () => { + expect(childFixturePaths('/snap/s', 0)).toEqual([]) + }) +}) + +describe('fixtureContext', () => { + it('reads the fixture header id and cwd', () => { + const ctx = fixtureContext('{"type":"session","id":"abc","cwd":"/rec"}\n{"type":"turn/start"}\n') + expect(ctx).toEqual({ sessionIds: ['abc'], cwd: '/rec' }) + }) + + it('yields no session ids for a header without a string id', () => { + expect(fixtureContext('{"type":"session","cwd":"/rec"}\n').sessionIds).toEqual([]) + }) + + it('falls back to an impossible sentinel cwd (never the empty string)', () => { + const ctx = fixtureContext('{"type":"session","id":"abc"}\n') + expect(ctx.cwd).toBe('\0no-cwd\0') + expect(ctx.cwd).not.toBe('') + }) + + it('treats an empty fixture as an empty header', () => { + expect(fixtureContext('')).toEqual({ sessionIds: [], cwd: '\0no-cwd\0' }) + }) +}) + +describe('normalizedHeaders', () => { + const header = (system: string): string => JSON.stringify({ + type: 'request/header', seq: 0, time: 9, data: { header: { config: { model: 'm' }, system }, reason: 'initial' }, + }) + + it('extracts every request/header payload in log order, normalized', () => { + const id = '11111111-2222-4333-8444-555555555555' + const log = `${JSON.stringify({ type: 'session', id, createdAt: 5, cwd: '/w' })}\n${header('one')}\n` + + `${JSON.stringify({ type: 'turn/start', seq: 1, time: 9, data: { turn: 1 } })}\n${header('two')}\n` + const headers = normalizedHeaders(log, { sessionIds: [id], cwd: '/w' }) + expect(headers).toEqual([ + { config: { model: 'm' }, system: 'one' }, + { config: { model: 'm' }, system: 'two' }, + ]) + }) + + it('yields nothing for a log without header events', () => { + const log = `${JSON.stringify({ type: 'session', id: 'a', createdAt: 5 })}\n` + expect(normalizedHeaders(log, { sessionIds: [], cwd: '/w' })).toEqual([]) + }) +}) + +describe('headerDeltaCount', () => { + it('counts request/header-delta events, ignoring blanks and other lines', () => { + const delta = JSON.stringify({ type: 'request/header-delta', seq: 2, time: 9, data: {} }) + const other = JSON.stringify({ type: 'request/header', seq: 0, time: 9, data: {} }) + expect(headerDeltaCount(`${other}\n\n${delta}\n${delta}\n`)).toBe(2) + expect(headerDeltaCount(`${other}\n`)).toBe(0) + }) +}) From b0144eaccdd85428d106a73075308f7c38f21167 Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 02:07:14 +0800 Subject: [PATCH 22/24] feat(acp-snapshot): scripted permission answers in the harness client MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit InputScript gains an optional permissionAnswers queue, consumed FIFO by the harness's requestPermission handler. Each entry selects by option KIND (allow_once, reject_once, …): option ids are agent-issued randoms a committed script cannot know, while kinds are the ACP-stable vocabulary, so the client maps kind → the offered optionId at answer time. An absent or exhausted queue answers cancelled — existing scenarios and goldens are untouched — and a scripted kind the request never offered throws, surfacing as a JSON-RPC error on the permission request: the scenario scripted an impossible click. This is what lets an approval-flow suite (the sandbox composition) drive allow/reject round-trips deterministically from input.json, per the shared-acp-snapshot RFC. --- packages/support/acp-snapshot/README.md | 2 +- packages/support/acp-snapshot/src/harness.ts | 36 ++++++++++++++++- packages/support/acp-snapshot/src/index.ts | 1 + .../acp-snapshot/tests/harness.spec.ts | 39 +++++++++++++++++++ 4 files changed, 75 insertions(+), 3 deletions(-) diff --git a/packages/support/acp-snapshot/README.md b/packages/support/acp-snapshot/README.md index 324cfa9f1d..a47a51d053 100644 --- a/packages/support/acp-snapshot/README.md +++ b/packages/support/acp-snapshot/README.md @@ -33,4 +33,4 @@ defineAcpSnapshotSuite({ The example also ships a `cordis.snapshot.yml` replay overlay next to its `cordis.yml` (the bin swaps them under `DSH_SNAPSHOT=replay` — [single-source replay config RFC](../../../docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md)); replay fixtures are served by [`dsh-llm-replay`](../llm-replay/README.md), which this package points at via the `DSH_SNAPSHOT_*` env vars it sets on the child. Fixture roles, record/replay semantics, and scenario-table fields are documented on `Scenario` and in the [snapshot RFC](../../../docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md). -Constraints: `suite.ts` imports vitest, so the package is importable only inside a vitest run (the harness and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the harness speaks the SDK's `ClientSideConnection` and answers `requestPermission` with `cancelled`. +Constraints: `suite.ts` imports vitest, so the package is importable only inside a vitest run (the harness and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the harness speaks the SDK's `ClientSideConnection`. Permission round-trips are scriptable: `InputScript.permissionAnswers` is a FIFO queue of option-kind selections (`allow_once`, `reject_once`, …) the client maps to the agent-issued `optionId` at answer time; an absent or exhausted queue answers `cancelled`, and a kind the request never offered fails loud. diff --git a/packages/support/acp-snapshot/src/harness.ts b/packages/support/acp-snapshot/src/harness.ts index 652e60a469..da640f2dce 100644 --- a/packages/support/acp-snapshot/src/harness.ts +++ b/packages/support/acp-snapshot/src/harness.ts @@ -89,6 +89,23 @@ export type InputStep = /** A scenario's `input.json`: an ordered list of input steps. */ export interface InputScript { steps: InputStep[] + /** + * Ordered answers for the agent's `session/request_permission` round-trips, + * consumed FIFO — the Nth request gets the Nth answer. Each answer selects + * by option KIND: option ids are agent-issued randoms a committed script + * cannot know, while kinds are the ACP-stable vocabulary, so the client maps + * kind → the offered `optionId` at answer time. A request beyond the queue + * (or with no queue at all) is answered `cancelled` — the stub behavior a + * scenario without approvals relies on. A scripted kind the request does + * not offer fails loud: the scenario scripted an impossible click. + */ + permissionAnswers?: PermissionAnswer[] +} + +/** One scripted answer to a permission request: which offered option kind to select. */ +export interface PermissionAnswer { + /** The `PermissionOption.kind` to select (`allow_once`, `reject_always`, …). */ + kind: 'allow_once' | 'allow_always' | 'reject_once' | 'reject_always' } /** One harvested session log plus the identifying facts off its header line. */ @@ -220,6 +237,9 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise const waitForUpdate = (match: (u: SessionNotification['update']) => boolean): Promise => new Promise(resolve => updateWaiters.push({ match, resolve })) + // Permission answers are consumed FIFO across the whole run; exhaustion + // falls back to `cancelled` so approval-free scenarios keep the plain stub. + const permissionQueue = [...input.permissionAnswers ?? []] const makeClient = (_agent: AcpAgent): Client => ({ sessionUpdate(params: SessionNotification): Promise { for (let i = updateWaiters.length - 1; i >= 0; i--) { @@ -236,8 +256,20 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise } return Promise.resolve() }, - requestPermission(_params: RequestPermissionRequest): Promise { - return Promise.resolve({ outcome: { outcome: 'cancelled' } }) + requestPermission(params: RequestPermissionRequest): Promise { + const answer = permissionQueue.shift() + if (answer === undefined) return Promise.resolve({ outcome: { outcome: 'cancelled' } }) + const option = params.options.find(o => o.kind === answer.kind) + if (option === undefined) { + // The scenario scripted a click the agent never offered — a scenario + // bug. Throwing here surfaces as a JSON-RPC error on the permission + // request, which the transcript (and usually the run) fails on. + throw new Error( + `snapshot-harness: scripted permission answer ${answer.kind} not among ` + + `the offered options [${params.options.map(o => o.kind).join(', ')}]`, + ) + } + return Promise.resolve({ outcome: { outcome: 'selected', optionId: option.optionId } }) }, }) const client = new ClientSideConnection(makeClient, stream) diff --git a/packages/support/acp-snapshot/src/index.ts b/packages/support/acp-snapshot/src/index.ts index a0d3380086..bbe74030f2 100644 --- a/packages/support/acp-snapshot/src/index.ts +++ b/packages/support/acp-snapshot/src/index.ts @@ -21,6 +21,7 @@ export { type HarvestedLog, type InputScript, type InputStep, + type PermissionAnswer, type RunOptions, type RunResult, } from './harness.ts' diff --git a/packages/support/acp-snapshot/tests/harness.spec.ts b/packages/support/acp-snapshot/tests/harness.spec.ts index 3884e36095..d674e01818 100644 --- a/packages/support/acp-snapshot/tests/harness.spec.ts +++ b/packages/support/acp-snapshot/tests/harness.spec.ts @@ -230,4 +230,43 @@ describe('runScenario', () => { ) expect(result.sessionLogs).toHaveLength(0) }) + + it('answers permission requests from the scripted queue by option kind, falling back to cancelled', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ permissionProbe: true }) + // Two prompts → two permission round-trips; one scripted answer, so the + // second request exercises the exhausted-queue fallback. + const result = await runScenario( + { + steps: [...boot, { op: 'prompt', text: 'one' }, { op: 'prompt', text: 'two' }], + permissionAnswers: [{ kind: 'allow_once' }], + }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + const first = result.rawStdout.indexOf('permission:{\\"outcome\\":\\"selected\\",\\"optionId\\":\\"opt-allow\\"}') + const second = result.rawStdout.indexOf('permission:{\\"outcome\\":\\"cancelled\\"}') + expect(first).toBeGreaterThanOrEqual(0) + expect(second).toBeGreaterThan(first) + }) + + it('selects a non-first offered option by kind', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ permissionProbe: true }) + const result = await runScenario( + { steps: [...boot, { op: 'prompt', text: 'deny it' }], permissionAnswers: [{ kind: 'reject_once' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.rawStdout).toContain('permission:{\\"outcome\\":\\"selected\\",\\"optionId\\":\\"opt-reject\\"}') + }) + + it('fails loud on a scripted permission kind the agent never offered', { timeout: 20_000 }, async () => { + const { fixtureFile } = await scenario({ permissionProbe: true }) + // The fake bin offers allow_once/reject_once; scripting allow_always is a + // scenario bug. The client handler throws, the SDK surfaces it as a + // JSON-RPC error on the permission request, and the fake bin echoes the + // missing outcome as null. + const result = await runScenario( + { steps: [...boot, { op: 'prompt', text: 'impossible click' }], permissionAnswers: [{ kind: 'allow_always' }] }, + { agent: AGENT, mode: 'replay', fixtureFile }, + ) + expect(result.rawStdout).toContain('permission:null') + }) }) From 9ab3a89ceacb2cb9f21689b7e18e0ca6d4e076a9 Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 02:09:32 +0800 Subject: [PATCH 23/24] docs(rfc): promote the shared-acp-snapshot RFC to implemented The package, coverage, and permission scripting all shipped on this branch, so the RFC moves to implemented/ with the lifecycle rewrite: Proposal becomes a present-tense Decision, Acceptance criteria and Risks fold into Testing/Consequences with what actually pinned each one (the zero-byte extraction parity, the 100% per-file coverage via the fake bin, the vitest-in-src caveat, the per-suite pin cost). --- docs/rfc/INDEX.md | 2 +- .../2026-07-08-shared-acp-snapshot-package.md | 36 +++++++++++++ .../2026-07-08-shared-acp-snapshot-package.md | 54 ------------------- 3 files changed, 37 insertions(+), 55 deletions(-) create mode 100644 docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md delete mode 100644 docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index 39927e3095..0d5f12d636 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -42,7 +42,6 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; |---|---| | [Deterministic tests, the replay invariant fixture, and race stress](proposed/testing/2026-06-11-deterministic-and-stress-testing.md) | 2026-06-11 | | [Mutation testing as the coverage counterweight](proposed/testing/2026-06-11-mutation-testing.md) | 2026-06-11 | -| [Extract the ACP snapshot suite into a support package](proposed/testing/2026-07-08-shared-acp-snapshot-package.md) | 2026-07-08 | ## Implemented @@ -165,6 +164,7 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; | [Hook snapshot matrix — end-to-end goldens for both bridges](implemented/testing/2026-07-04-hook-snapshot-matrix.md) | 2026-07-04 | | [Single-source the acp-agent replay config](implemented/testing/2026-07-04-single-source-acp-replay-config.md) | 2026-07-04 | | [Pin request-header content in one snapshot scenario](implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md) | 2026-07-06 | +| [Extract the ACP snapshot suite into a support package](implemented/testing/2026-07-08-shared-acp-snapshot-package.md) | 2026-07-08 | ## Rejected diff --git a/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md new file mode 100644 index 0000000000..8c95c1f049 --- /dev/null +++ b/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md @@ -0,0 +1,36 @@ +# RFC: Extract the ACP snapshot suite into a support package + +Status: implemented + +## Problem + +The ACP snapshot tier ([snapshot RFC](2026-06-19-acp-snapshot-tests.md)) was built from three modules living inside one example's test directory: `snapshot-harness.ts` (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), `snapshot-normalize.ts` (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in `acp.snapshot.ts` (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). + +A second ACP example wanting snapshot coverage — the sandbox/approval composition is the immediate consumer — could only copy those modules, forking exactly the logic that must not drift: record write-back, header scrubbing, child-session harvest ordering. The spawn/client glue was already triplicated across `acp.e2e.ts`, `hooks.e2e.ts`, and the harness (`TODO(acp-test-harness)`). Location also decided test rigor: the per-file 100% coverage gate measures `packages/*/*/src` only, so none of this machinery was measured — the same gap that had moved `dsh-llm-replay` out of `examples/` into [packages/support](../../../../packages/support/README.md). And the harness's ACP client hardcoded `requestPermission → cancelled`, so an approval round-trip — the headline behavior of the sandbox composition — could not be expressed at the snapshot tier at all. + +## Decision + +The machinery lives in [`packages/support/acp-snapshot`](../../../../packages/support/acp-snapshot/README.md) (`@deepseek-ai/dsh-acp-snapshot`); an example's `*.snapshot.ts` is its scenario table, its agent paths, and one factory call, over its own `snapshots/` fixtures and `cordis.snapshot.yml` overlay ([single-source replay config](2026-07-04-single-source-acp-replay-config.md)). Reading `DSH_SNAPSHOT` stays at that edge — the library takes a resolved `mode`. + +**`src/harness.ts`** — `runScenario` and the input-script/result types, parameterized by an `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath`; absolute paths the consuming suite resolves from its own `import.meta.url`). The client's `session/request_permission` handler consumes an optional `InputScript.permissionAnswers` FIFO queue, each entry selecting by option **kind** (ids are agent-issued randoms a committed script cannot know; kinds are the ACP-stable vocabulary, mapped to the offered `optionId` at answer time); an absent or exhausted queue answers `cancelled`, and a kind the request never offered fails loud. This is what lets an approval suite drive allow/reject round-trips deterministically from `input.json`. + +**`src/normalize.ts`** — the pure normalizers, hook-free by policy: when a future event carries a new volatile field (an approval duration, say), the shared normalizer learns it in the same change, keeping one home for what "normalized" means rather than per-suite scrub extensions. + +**`src/suite.ts`** — the `Scenario` type and `defineAcpSnapshotSuite(options)`, registering the per-scenario compares, record-mode fixture write-back, the header pin with its live uniformity guard, and the fixture guard block (no orphan scenario dirs, required files present, exactly one pin, non-pinning fixtures are `scrubRequestHeaders` fixed points). The pinned-header contract ([pinned-header RFC](2026-07-06-pin-request-header-content-in-one-scenario.md)) is per-suite: each suite flags exactly one `pinsHeader` scenario (the factory throws on zero, a meta-test rejects more than one; WHICH scenario pins is the table's reviewable choice), and the uniformity guard compares only that suite's sessions. The pure helpers (`childFixturePaths`, `fixtureContext`, `normalizedHeaders`, `headerDeltaCount`) are exported for direct unit coverage. + +## Alternatives considered + +- **Copy the modules into each example** — the fork this RFC exists to prevent: the record/guard logic is exactly the code that must stay byte-identical across suites, and examples are outside the coverage gate, so each copy is also unmeasured. +- **A shared module directory under `examples/`** — keeps the code outside the coverage gate and forces relative imports across example boundaries, against the package-name import convention; `examples/` leaves stay thin by design. +- **A `/testing` subpath export of `dsh-acp-agent`** — couples test infrastructure into a product package's surface and dependency set; `packages/support/` exists precisely for real-but-lower-compatibility dev/test packages, with `dsh-llm-replay` as the precedent this package completes. +- **Export raw test-body functions instead of a suite factory** — each example would re-own the `describe`/`it` skeleton (~80 lines of registration boilerplate per suite) for no flexibility gain; the factory keeps consumers to a scenario table plus one call, and the exported pure helpers preserve unit-testability inside the factory design. +- **An injectable ACP `Client` factory instead of declarative `permissionAnswers`** — maximally flexible, but it leaks SDK client construction to every consumer and reopens per-example drift in exactly the layer being unified; a declarative queue keeps `input.json` the single scripting surface and stays golden-normalizable. +- **Generalize beyond ACP (a transport-agnostic snapshot harness)** — no second transport exists; the harness is ACP-shaped end to end (SDK client, JSON-RPC frames, `session/update` waiters), and a speculative abstraction would be a seam split ahead of any consumer. + +## Testing + +Extraction parity was proven mechanically: after the move, `pnpm run test:snapshot` matched the base commit's result with zero byte changes under `examples/acp-agent/tests/snapshots/`. The package's `src/` holds per-file 100% statements/branches/functions/lines under the gating unit run, driven through the REAL spawn path by a scripted fake ACP bin (`tests/fixtures/fake-acp-agent.ts`, behavior scripted per scenario via a `behavior.json` beside the fixture): `harness.spec.ts` covers every step op, both expect-error arms, the permission queue (selection, fallback, impossible-click), env forwarding, workspace seeding, and the harvest ordering/noise/fallback branches; `suite.spec.ts` runs the factory for real at collection time — a replay suite over committed synthetic fixtures and a record suite over a temp copy (write-back never touches the committed tree; `ACP_SNAPSHOT_SPEC_BOOTSTRAP=1` re-bootstraps it) — plus direct cases for the pure helpers. Two structurally unreachable guards carry reasoned `v8 ignore` comments. The fake bin substitutes the `session/new` cwd, not `process.cwd()`, into scripted logs, matching what the real bin's header carries (darwin realpaths `/var/folders/…` to `/private/var/folders/…`). + +## Consequences + +A new example gets the whole snapshot tier from a scenario table plus fixtures — the sandbox branch merges master down and adds its own suite (own pin scenario, own overlay, fixtures via `test:snapshot:record`, approvals via `permissionAnswers`). The costs: `suite.ts` imports vitest, so the package is importable only inside a vitest run — a shape no other package has, stated in its README; each suite pins its own ~8 KB header fixture (a genuinely distinct composition deserves its own pin; an identical one would be caught by that suite's uniformity guard); and the e2e launcher duplication remains (`TODO(acp-test-harness)`) — the harness is the extraction target when that migration lands. diff --git a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md deleted file mode 100644 index 839e1d4542..0000000000 --- a/docs/rfc/proposed/testing/2026-07-08-shared-acp-snapshot-package.md +++ /dev/null @@ -1,54 +0,0 @@ -# RFC: Extract the ACP snapshot suite into a support package - -Status: proposed - -## Problem - -The ACP snapshot tier ([snapshot RFC](../../implemented/testing/2026-06-19-acp-snapshot-tests.md)) is built from three modules that live inside one example's test directory: `snapshot-harness.ts` (boot the real bin subprocess, drive it over ACP JSON-RPC, harvest the persisted logs), `snapshot-normalize.ts` (the pure golden normalizers), and the ~150-line scenario body plus fixture guards in [acp.snapshot.ts](../../../../examples/acp-agent/tests/acp.snapshot.ts) (record/replay modes, the stdout-golden and log compares, the pinned-header uniformity guard, the orphan/required-file/single-pin meta-tests). - -A second ACP example that wants snapshot coverage — the sandbox/approval composition is the immediate consumer — can only copy those modules, forking exactly the logic that must not drift: record write-back, header scrubbing, child-session harvest ordering. The spawn/client glue is already triplicated across [acp.e2e.ts](../../../../examples/acp-agent/tests/acp.e2e.ts), [hooks.e2e.ts](../../../../examples/acp-agent/tests/hooks.e2e.ts), and the harness, marked by `TODO(acp-test-harness)`. - -Location also decides test rigor: the per-file 100% coverage gate measures `packages/*/*/src` only, so none of this machinery is measured — the same gap that moved `dsh-llm-replay` out of `examples/` into [packages/support](../../../../packages/support/README.md). The harness's subprocess lifecycle, teardown, and harvest-ordering branches are exercised only transitively, when a live scenario happens to hit them. - -Finally, the harness's ACP client hardcodes `requestPermission → cancelled`, so an approval round-trip — the headline behavior of the sandbox composition — cannot be expressed at the snapshot tier at all. A new transcript surface must name its coverage at every tier at plan time; today the tier cannot express this one. - -## Proposal - -Create `packages/support/acp-snapshot` (`@deepseek-ai/dsh-acp-snapshot`), a support-tier package with three source modules; each example keeps only its scenario table, its `snapshots/` fixtures, its `cordis.snapshot.yml` overlay ([single-source replay config](../../implemented/testing/2026-07-04-single-source-acp-replay-config.md)), and the paths that identify its agent. - -**`src/harness.ts`** — `runScenario` and the input-script/result types, moved intact, with the module-level path constants replaced by an explicit `AgentUnderTest` parameter (`binScript`, `configPath`, `tsconfigPath`): defaulting stays at the seam's consumer, which resolves them from its own `import.meta.url`. The internal spawn/tee/SDK-client wiring is factored so the e2e launcher duplication can migrate onto it later; that migration is out of scope here and the `TODO(acp-test-harness)` stays until it lands. - -**`src/normalize.ts`** — the normalizers move verbatim with their spec. They stay hook-free: when a future event carries a new volatile field (an approval duration, say), the shared normalizer learns it in the same change, keeping one home for what "normalized" means rather than per-suite scrub extensions. - -**`src/suite.ts`** — the `Scenario` type and `defineAcpSnapshotSuite(options)`, which registers the per-scenario `describe`/`it` tree and the fixture guard tests. Options carry the resolved `mode: 'replay' | 'record'` — reading `DSH_SNAPSHOT` stays at the edge, in the example's `*.snapshot.ts`. The guard logic (no orphan scenario dirs, required fixture files, exactly one pin, non-pinning fixtures are `scrubRequestHeaders` fixed points) is exported as pure assertion functions the registered tests call one-line-each, so failure paths are unit-testable without meta-running vitest. The pinned-header contract ([pinned-header RFC](../../implemented/testing/2026-07-06-pin-request-header-content-in-one-scenario.md)) becomes per-suite: each suite flags exactly one `pinsHeader` scenario and its uniformity guard compares only that suite's sessions, which is the guard's existing scope. - -**Scripted permission answers** — `InputScript` gains an optional ordered `permissionAnswers` queue consumed by the harness client's `requestPermission`, each entry selecting a response by option **kind** (`allow_once`, `reject_once`, …); the harness maps kind to the agent-issued `optionId` at answer time, since ids are random per run while kinds are stable. An exhausted or absent queue falls back to today's `cancelled`, so existing scenarios and goldens are untouched. This is what lets a sandbox suite script an approval round-trip deterministically from `input.json`. - -Repo wiring follows the [adding-a-package cookbook](../../../cookbook/adding-a-package.md): manifest per the workspace constraints (cordis peer+dev, `private`, standard `files`), references in the root and build tsconfigs (the `@deepseek-ai/dsh-*` paths wildcard already covers `packages/support/*/src`), a row in the support group README, and explicit `@agentclientprotocol/sdk`/`vitest`/`tsx` dependencies instead of inherited-by-walk-up resolution. [docs/testing.md](../../../testing.md) generalizes "scenarios live under `examples/acp-agent/tests/snapshots/`" to the owning example's `tests/snapshots/`. - -Landing order is three commits on one PR: (1) the pure move plus parameterization, with `examples/acp-agent/tests/acp.snapshot.ts` collapsed to its scenario table and one `defineAcpSnapshotSuite` call; (2) the coverage work — a scripted fake ACP bin fixture (reads JSON-RPC frames on stdin, emits canned responses and session updates, writes synthetic session JSONL under `DSH_SNAPSHOT_SESSIONS_ROOT`) driving `harness.ts` through every step op, expect-error branch, child-harvest ordering, and teardown path, and `suite.spec.ts` registering synthetic replay- and record-mode suites against temp fixture dirs (record mode is keyless here: the live API sits behind the bin, and the fake bin needs none); (3) `permissionAnswers` with its unit coverage. The sandbox branch then merges master down and adds its own suite: scenario table, own pin scenario, own overlay, fixtures recorded via `test:snapshot:record`. - -## Alternatives considered - -- **Copy the modules into each example** — the fork this RFC exists to prevent: the record/guard logic is exactly the code that must stay byte-identical across suites, and examples are outside the coverage gate, so each copy is also unmeasured. -- **A shared module directory under `examples/`** — keeps the code outside the coverage gate and forces relative imports across example boundaries, against the package-name import convention; `examples/` leaves stay thin by design. -- **A `/testing` subpath export of `dsh-acp-agent`** — couples test infrastructure into a product package's surface and dependency set; `packages/support/` exists precisely for real-but-lower-compatibility dev/test packages, with `dsh-llm-replay` as the precedent this proposal completes. -- **Export raw test-body functions instead of a suite factory** — each example would re-own the `describe`/`it` skeleton (~80 lines of registration boilerplate per suite) for no flexibility gain; the factory keeps consumers to a scenario table plus one call, and the pure guard functions preserve unit-testability inside the factory design. -- **An injectable ACP `Client` factory instead of declarative `permissionAnswers`** — maximally flexible, but it leaks SDK client construction to every consumer and reopens per-example drift in exactly the layer being unified; a declarative queue keeps `input.json` the single scripting surface and stays golden-normalizable. -- **Generalize beyond ACP (a transport-agnostic snapshot harness)** — no second transport exists; the harness is ACP-shaped end to end (SDK client, JSON-RPC frames, `session/update` waiters), and a speculative abstraction would be a seam split ahead of any consumer. - -## Acceptance criteria - -- After the pure-move commit, `pnpm run test:snapshot` is green with zero byte changes under `examples/acp-agent/tests/snapshots/` — the machine proof that extraction changed no behavior. -- `pnpm run test:coverage` holds the new package's `src/` at per-file 100% with keyless unit specs; any `v8 ignore` carries its reason. -- `examples/acp-agent/tests/acp.snapshot.ts` contains no golden/compare/guard logic — only the scenario table, the agent paths, and the factory call. -- A harness unit test drives a `permissionAnswers` script through kind→`optionId` mapping and the exhaustion fallback, demonstrating the tier can express an approval round-trip before the sandbox suite needs it. -- `doc-sync`, `hygiene`, and `verify-module-graph` pass with the new package wired in. - -## Risks - -- **Per-file 100% on `suite.ts`** is the tightest constraint: `toMatchFileSnapshot` update semantics differ under CI, and factory-registered tests must be driven by real vitest collection. The pure-guard-function split plus synthetic-suite registration is the mitigation; a justified `v8 ignore` is the last resort, not the plan. -- **`vitest` becomes a `src` dependency** of a workspace package (the factory imports `describe`/`it`/`expect`), so importing `suite.ts` outside a vitest run throws — acceptable for a support-tier package and stated in its README, but it is a shape no other package has. -- **Fake-bin drift**: harness unit tests exercise plumbing against a scripted bin, not the real one. The real bin path stays exercised on every `test:snapshot` run, so drift surfaces there; the fake bin only owns branches the live suite cannot deterministically reach. -- **Per-suite pins duplicate header bulk**: each new suite commits one full ~8 KB header fixture. Accepted — a suite whose composition equals another's is the degenerate case the uniformity guard would surface, and one pinned line per genuinely distinct composition is the pinned-header design applied at its natural scope. -- The extraction touches the gating snapshot suite itself; a subtle behavior change would surface as golden churn. The zero-byte-diff acceptance criterion is the guard. From 1097fa3507e63afc179392cebb84f56ddedfd83b Mon Sep 17 00:00:00 2001 From: kingwl Date: Wed, 8 Jul 2026 02:46:23 +0800 Subject: [PATCH 24/24] fix review finding: an impossible scripted permission click rejects the run MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A client-callback throw only becomes a JSON-RPC error RESPONSE to the agent's session/request_permission — runScenario itself kept going, so a tolerant agent could treat the error as a denial and the scenario would pass, or worse, record: the impossible click baked into fixture and golden, green on every replay. The mismatch is now captured as a harness error while the agent is answered plain cancelled (a well-defined path it cannot reinterpret), and the step loop rejects the run on it as soon as the in-flight step settles. The spec asserts the rejection instead of the agent-side error echo. --- .../2026-07-08-shared-acp-snapshot-package.md | 2 +- packages/support/acp-snapshot/README.md | 2 +- packages/support/acp-snapshot/src/harness.ts | 24 +++++++++++++++---- .../acp-snapshot/tests/harness.spec.ts | 14 +++++------ 4 files changed, 29 insertions(+), 13 deletions(-) diff --git a/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md b/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md index 8c95c1f049..910f179313 100644 --- a/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md +++ b/docs/rfc/implemented/testing/2026-07-08-shared-acp-snapshot-package.md @@ -12,7 +12,7 @@ A second ACP example wanting snapshot coverage — the sandbox/approval composit The machinery lives in [`packages/support/acp-snapshot`](../../../../packages/support/acp-snapshot/README.md) (`@deepseek-ai/dsh-acp-snapshot`); an example's `*.snapshot.ts` is its scenario table, its agent paths, and one factory call, over its own `snapshots/` fixtures and `cordis.snapshot.yml` overlay ([single-source replay config](2026-07-04-single-source-acp-replay-config.md)). Reading `DSH_SNAPSHOT` stays at that edge — the library takes a resolved `mode`. -**`src/harness.ts`** — `runScenario` and the input-script/result types, parameterized by an `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath`; absolute paths the consuming suite resolves from its own `import.meta.url`). The client's `session/request_permission` handler consumes an optional `InputScript.permissionAnswers` FIFO queue, each entry selecting by option **kind** (ids are agent-issued randoms a committed script cannot know; kinds are the ACP-stable vocabulary, mapped to the offered `optionId` at answer time); an absent or exhausted queue answers `cancelled`, and a kind the request never offered fails loud. This is what lets an approval suite drive allow/reject round-trips deterministically from `input.json`. +**`src/harness.ts`** — `runScenario` and the input-script/result types, parameterized by an `AgentUnderTest` (`binScript`, `configPath`, `tsconfigPath`; absolute paths the consuming suite resolves from its own `import.meta.url`). The client's `session/request_permission` handler consumes an optional `InputScript.permissionAnswers` FIFO queue, each entry selecting by option **kind** (ids are agent-issued randoms a committed script cannot know; kinds are the ACP-stable vocabulary, mapped to the offered `optionId` at answer time); an absent or exhausted queue answers `cancelled`, and a kind the request never offered rejects the run — the agent itself is answered `cancelled`, so the scenario bug fails the harness rather than being absorbed as an agent-side denial. This is what lets an approval suite drive allow/reject round-trips deterministically from `input.json`. **`src/normalize.ts`** — the pure normalizers, hook-free by policy: when a future event carries a new volatile field (an approval duration, say), the shared normalizer learns it in the same change, keeping one home for what "normalized" means rather than per-suite scrub extensions. diff --git a/packages/support/acp-snapshot/README.md b/packages/support/acp-snapshot/README.md index a47a51d053..8c0b514c07 100644 --- a/packages/support/acp-snapshot/README.md +++ b/packages/support/acp-snapshot/README.md @@ -33,4 +33,4 @@ defineAcpSnapshotSuite({ The example also ships a `cordis.snapshot.yml` replay overlay next to its `cordis.yml` (the bin swaps them under `DSH_SNAPSHOT=replay` — [single-source replay config RFC](../../../docs/rfc/implemented/testing/2026-07-04-single-source-acp-replay-config.md)); replay fixtures are served by [`dsh-llm-replay`](../llm-replay/README.md), which this package points at via the `DSH_SNAPSHOT_*` env vars it sets on the child. Fixture roles, record/replay semantics, and scenario-table fields are documented on `Scenario` and in the [snapshot RFC](../../../docs/rfc/implemented/testing/2026-06-19-acp-snapshot-tests.md). -Constraints: `suite.ts` imports vitest, so the package is importable only inside a vitest run (the harness and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the harness speaks the SDK's `ClientSideConnection`. Permission round-trips are scriptable: `InputScript.permissionAnswers` is a FIFO queue of option-kind selections (`allow_once`, `reject_once`, …) the client maps to the agent-issued `optionId` at answer time; an absent or exhausted queue answers `cancelled`, and a kind the request never offered fails loud. +Constraints: `suite.ts` imports vitest, so the package is importable only inside a vitest run (the harness and normalizers have no such dependency but ship from the same entry). ACP-specific by design — the harness speaks the SDK's `ClientSideConnection`. Permission round-trips are scriptable: `InputScript.permissionAnswers` is a FIFO queue of option-kind selections (`allow_once`, `reject_once`, …) the client maps to the agent-issued `optionId` at answer time; an absent or exhausted queue answers `cancelled`, and a kind the request never offered rejects the run (the agent is answered `cancelled`, so a tolerant agent cannot absorb the scenario bug). diff --git a/packages/support/acp-snapshot/src/harness.ts b/packages/support/acp-snapshot/src/harness.ts index da640f2dce..538b81d57e 100644 --- a/packages/support/acp-snapshot/src/harness.ts +++ b/packages/support/acp-snapshot/src/harness.ts @@ -97,7 +97,9 @@ export interface InputScript { * kind → the offered `optionId` at answer time. A request beyond the queue * (or with no queue at all) is answered `cancelled` — the stub behavior a * scenario without approvals relies on. A scripted kind the request does - * not offer fails loud: the scenario scripted an impossible click. + * not offer REJECTS the run: the scenario scripted an impossible click, + * and {@link runScenario} throws once the in-flight step settles (the + * agent itself just sees `cancelled`, so it cannot absorb the bug). */ permissionAnswers?: PermissionAnswer[] } @@ -240,6 +242,14 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise // Permission answers are consumed FIFO across the whole run; exhaustion // falls back to `cancelled` so approval-free scenarios keep the plain stub. const permissionQueue = [...input.permissionAnswers ?? []] + // A scenario bug detected inside a client callback (a scripted permission + // kind the agent never offered). It cannot fail the run from in there: a + // callback throw only becomes a JSON-RPC error RESPONSE to the agent, and + // a tolerant agent treats that as a denial and carries on — the run (or + // worse, a record) would absorb the impossible click silently. So the + // callback answers `cancelled` (a well-defined path for the agent), + // captures the error here, and the step loop fails the run on it. + let scriptError: Error | undefined const makeClient = (_agent: AcpAgent): Client => ({ sessionUpdate(params: SessionNotification): Promise { for (let i = updateWaiters.length - 1; i >= 0; i--) { @@ -262,12 +272,13 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise const option = params.options.find(o => o.kind === answer.kind) if (option === undefined) { // The scenario scripted a click the agent never offered — a scenario - // bug. Throwing here surfaces as a JSON-RPC error on the permission - // request, which the transcript (and usually the run) fails on. - throw new Error( + // bug. Captured (last one wins; same bug class either way) and + // answered `cancelled`; the step loop rejects the run on it. + scriptError = new Error( `snapshot-harness: scripted permission answer ${answer.kind} not among ` + `the offered options [${params.options.map(o => o.kind).join(', ')}]`, ) + return Promise.resolve({ outcome: { outcome: 'cancelled' } }) } return Promise.resolve({ outcome: { outcome: 'selected', optionId: option.optionId } }) }, @@ -276,6 +287,11 @@ export async function runScenario(input: InputScript, opts: RunOptions): Promise for (const step of input.steps) { await runStep(client, step, cwd, waitForUpdate, () => sessionId, (id) => { sessionId = id }) + // A permission exchange happens while a step's request is in flight, so + // by the time the step settles any script bug it exposed is captured — + // fail the run HERE, as a harness error, rather than hoping the agent's + // reaction to the answer perturbs the transcript. + if (scriptError !== undefined) throw scriptError } // Done driving: close stdin so the server disposes gracefully (flushing // persistence) and exits. Then await exit so the harvested log is complete. diff --git a/packages/support/acp-snapshot/tests/harness.spec.ts b/packages/support/acp-snapshot/tests/harness.spec.ts index d674e01818..683d2aaf80 100644 --- a/packages/support/acp-snapshot/tests/harness.spec.ts +++ b/packages/support/acp-snapshot/tests/harness.spec.ts @@ -257,16 +257,16 @@ describe('runScenario', () => { expect(result.rawStdout).toContain('permission:{\\"outcome\\":\\"selected\\",\\"optionId\\":\\"opt-reject\\"}') }) - it('fails loud on a scripted permission kind the agent never offered', { timeout: 20_000 }, async () => { + it('rejects the run on a scripted permission kind the agent never offered', { timeout: 20_000 }, async () => { const { fixtureFile } = await scenario({ permissionProbe: true }) // The fake bin offers allow_once/reject_once; scripting allow_always is a - // scenario bug. The client handler throws, the SDK surfaces it as a - // JSON-RPC error on the permission request, and the fake bin echoes the - // missing outcome as null. - const result = await runScenario( + // scenario bug. The agent is answered `cancelled` (it must not be able to + // absorb the bug as an error-means-denial), and the RUN fails: a callback + // throw would only reach the agent as a JSON-RPC error response, letting + // a tolerant agent carry on and the scenario pass — or record. + await expect(runScenario( { steps: [...boot, { op: 'prompt', text: 'impossible click' }], permissionAnswers: [{ kind: 'allow_always' }] }, { agent: AGENT, mode: 'replay', fixtureFile }, - ) - expect(result.rawStdout).toContain('permission:null') + )).rejects.toThrow(/allow_always not among the offered options \[allow_once, reject_once\]/) }) })