Merge origin/master into timeout-design
Resolve conflicts from master's catalog/doc refactors landing alongside the tool-call timeout work: - knip.json: keep both new workspace entries (util/timeout + support/acp-snapshot). - tool-web/src/fetch.ts: keep the timeout_ms removal, adopt master's richer JSDoc @param/@returns style on parseFetchArgs/presentFetchCall. - tools/README.md: keep the tools/execute pipeline wording, adopt master's flattened docs/tool-catalog.md path. - Regenerate every generated doc (cordis-catalog, tool-catalog, config-catalog, doc-graphs, module-graph) so they carry both master's changes and the tools/execute event + timeout-policy package. - Add @param/@returns to toolTimeoutResult for master's new verify-export-jsdoc gate.
This commit is contained in:
@@ -62,6 +62,8 @@ export const SENSITIVE_ENV_PATTERN = /KEY|SECRET|TOKEN/i
|
||||
* by in-process plugins (the hooks bridges), not the model — `dsh-tool-bash`
|
||||
* builds its request from named fields only and does not forward model input
|
||||
* here (see its README, § "The tool builds its request from named args only").
|
||||
* @param extra - caller-supplied entries merged last; an explicit entry wins even against the scrub and the overrides.
|
||||
* @returns the environment to hand to `spawn` for the child process.
|
||||
*/
|
||||
export function childEnv(extra?: Record<string, string>): NodeJS.ProcessEnv {
|
||||
const env: NodeJS.ProcessEnv = {}
|
||||
@@ -160,6 +162,14 @@ export class OutputCollector {
|
||||
private readonly spillDir: string,
|
||||
) {}
|
||||
|
||||
/**
|
||||
* Ingest one stream chunk, counting it toward the whole-stream total. On
|
||||
* first overflow of the in-memory cap a spill file is opened and every chunk
|
||||
* (already-collected ones included) is appended there from then on; the
|
||||
* in-memory tail then drops whole chunks from its head (or the head of a
|
||||
* single over-cap chunk) until it fits the cap again.
|
||||
* @param chunk - the raw bytes from one stream 'data' event.
|
||||
*/
|
||||
push(chunk: Buffer): void {
|
||||
this.total += chunk.length
|
||||
const overflows = this.bytes + chunk.length > this.maxBytes
|
||||
@@ -203,7 +213,10 @@ export class OutputCollector {
|
||||
// the bottom of this file) and `totalBytes` is read only by a test. The live
|
||||
// background-poll path goes through `readFrom()`, so inline snapshot() into
|
||||
// finalize() and drop or privatize the totalBytes getter.
|
||||
/** Read the collected tail without finalizing (the final-result snapshot). */
|
||||
/**
|
||||
* Read the collected tail without finalizing (the final-result snapshot).
|
||||
* @returns the retained tail text, the truncation flag, and the spill path when one was created.
|
||||
*/
|
||||
snapshot(): CollectedOutput {
|
||||
return {
|
||||
text: Buffer.concat(this.chunks).toString('utf8'),
|
||||
@@ -222,6 +235,8 @@ export class OutputCollector {
|
||||
* pushed since `fromByte`. When `fromByte` has already slid out of the
|
||||
* in-memory tail window, the read is `lossy` — it returns the whole
|
||||
* retained tail and the gap is only recoverable from the spill file.
|
||||
* @param fromByte - whole-stream offset to resume from (a prior read's `nextOffset`; 0 for the first read).
|
||||
* @returns the delta text, the offset for the next read, the `lossy` flag, and the spill path when one was created.
|
||||
*/
|
||||
readFrom(fromByte: number): { text: string; nextOffset: number; lossy: boolean; spillPath?: string } {
|
||||
const windowStart = this.total - this.bytes
|
||||
@@ -236,7 +251,12 @@ export class OutputCollector {
|
||||
}
|
||||
}
|
||||
|
||||
/** Close the spill file (if any) and return the final output. */
|
||||
/**
|
||||
* Close the spill file (if any) and return the final output. A failed close
|
||||
* (delayed writeback fault) stops advertising the spill path — the file may
|
||||
* be missing its tail — but still returns the in-memory result.
|
||||
* @returns the final collected output: tail text, truncation flag, and the spill path when intact.
|
||||
*/
|
||||
finalize(): CollectedOutput {
|
||||
if (this.spillFd !== undefined) {
|
||||
try {
|
||||
@@ -262,6 +282,8 @@ export class OutputCollector {
|
||||
* host process — a kill that cannot be delivered is reported by the process
|
||||
* NOT dying, which callers already handle via escalation/timeouts. No-op for
|
||||
* non-positive pids (spawn never started a process).
|
||||
* @param pid - the group leader's pid; non-positive means the spawn failed and the call is a no-op.
|
||||
* @param sig - the signal to deliver to the whole group.
|
||||
*/
|
||||
export function killGroup(pid: number, sig: NodeJS.Signals): void {
|
||||
if (pid <= 0) return
|
||||
@@ -303,6 +325,9 @@ export interface RunningBash {
|
||||
* exec sessions addressable via session ids + stdin writes. We deliberately
|
||||
* spawn a fresh non-login `bash -c` per call for determinism (no rc files,
|
||||
* no inherited shell state); revisit when real workflows demand it.
|
||||
* @param spec - the fully-resolved run (command, cwd, limits); no defaulting happens here.
|
||||
* @param internals - test-only knobs; omitted fields fall back to the private per-process spill dir.
|
||||
* @returns the live handle: pid, the two live collectors, the outcome promise, and `kill()`.
|
||||
*/
|
||||
export function runBash(spec: SpawnSpec, internals: RunInternals = {}): RunningBash {
|
||||
const spillDir = internals.spillDir ?? privateSpillDir()
|
||||
|
||||
@@ -11,7 +11,11 @@ import type { Branded } from '@deepseek-ai/dsh-brand'
|
||||
/** Identifies one background task within an executor (generated `bash-N`). */
|
||||
export type BashTaskId = Branded<'BashTaskId'>
|
||||
|
||||
/** Brand a string as a {@link BashTaskId}. */
|
||||
/**
|
||||
* Brand a string as a {@link BashTaskId}.
|
||||
* @param id - the raw task-id string (the executor generates `bash-N`).
|
||||
* @returns the same string, branded; no validation is performed.
|
||||
*/
|
||||
export function BashTaskId(id: string): BashTaskId {
|
||||
return id as BashTaskId
|
||||
}
|
||||
@@ -26,7 +30,12 @@ export function BashTaskId(id: string): BashTaskId {
|
||||
*/
|
||||
export type OwnerToken = Branded<'OwnerToken'>
|
||||
|
||||
/** Brand a string as an {@link OwnerToken}. */
|
||||
/**
|
||||
* Brand a string as an {@link OwnerToken}. Only the consuming boundary
|
||||
* (`dsh-tool-bash`) should cast its own id vocabulary in — see the type's doc.
|
||||
* @param id - the consumer's raw owner identity (the tool layer passes the owning agent's session id).
|
||||
* @returns the same string, branded; no validation is performed.
|
||||
*/
|
||||
export function OwnerToken(id: string): OwnerToken {
|
||||
return id as OwnerToken
|
||||
}
|
||||
|
||||
@@ -99,6 +99,8 @@ function streamText(output: CollectedOutput): string {
|
||||
* stderr section, then exit-status markers. Non-zero exits are REPORTED, not
|
||||
* errored — the model decides how to react; only infrastructure failures
|
||||
* (spawn errors, aborts) surface as isError results.
|
||||
* @param result - the completed foreground run from the executor.
|
||||
* @returns the model-facing text: output body (or `(no output)`), then any timeout/signal/exit markers, each on its own line.
|
||||
*/
|
||||
export function renderResult(result: BashRunResult): string {
|
||||
const out = streamText(result.stdout)
|
||||
|
||||
@@ -216,6 +216,11 @@ export class BasicCompactService extends CompactService {
|
||||
* Estimate the token count of content blocks — chars divided by the
|
||||
* `charsPerToken` config, with per-block overhead. Override in a subclass to
|
||||
* plug in a real tokenizer.
|
||||
*
|
||||
* @param blocks - the blocks to estimate; `tool-result` blocks recurse into
|
||||
* their nested content, and unknown (merge-extended) types fall back to
|
||||
* their JSON-stringified length.
|
||||
* @returns the estimated token count.
|
||||
*/
|
||||
estimateContentTokens(blocks: readonly ContentBlock[]): number {
|
||||
const { charsPerToken } = this.config
|
||||
@@ -246,6 +251,11 @@ export class BasicCompactService extends CompactService {
|
||||
/**
|
||||
* Estimate token count for a single session event. Returns 0 for non-message
|
||||
* event types (boundaries, chunks, usage, errors, compact markers).
|
||||
*
|
||||
* @param event - any session event; only the message-bearing types carry
|
||||
* content to count.
|
||||
* @returns the estimated token count of the event's content, or 0 for a
|
||||
* non-message event.
|
||||
*/
|
||||
estimateEventTokens(event: SessionEvent): number {
|
||||
switch (event.type) {
|
||||
@@ -260,7 +270,14 @@ export class BasicCompactService extends CompactService {
|
||||
}
|
||||
}
|
||||
|
||||
/** Estimate total tokens across a list of messages plus optional system prompt. */
|
||||
/**
|
||||
* Estimate total tokens across a list of messages plus optional system prompt.
|
||||
*
|
||||
* @param messages - the derived conversation messages; each adds a fixed
|
||||
* role-framing overhead on top of its content estimate.
|
||||
* @param systemPrompt - counted at chars / `charsPerToken` when provided.
|
||||
* @returns the estimated token footprint of the whole request.
|
||||
*/
|
||||
estimateTokens(messages: readonly Message[], systemPrompt?: string): number {
|
||||
let total = 0
|
||||
for (const msg of messages) {
|
||||
@@ -293,6 +310,13 @@ export class BasicCompactService extends CompactService {
|
||||
* used (`model`, `maxTokens`) — the caller logs the envelope on the
|
||||
* `compact/summary` provenance event, so an overriding subclass (template
|
||||
* or remote summarizer) reports its own envelope honestly.
|
||||
*
|
||||
* @param text - plain-text rendering of the conversation region to condense.
|
||||
* @param agent - supplies the fallback model and the session id stamped on
|
||||
* the call; throws when neither it nor the config names a model.
|
||||
* @param signal - optional abort signal, forwarded into the model call.
|
||||
* @returns the text-only summary blocks plus the call envelope used
|
||||
* (`model`, and `maxTokens` when the summarizer has a cap).
|
||||
*/
|
||||
async summarize(
|
||||
text: string, agent: Agent, signal?: AbortSignal,
|
||||
|
||||
@@ -54,6 +54,9 @@ export type ResolvedConfig = Required<BasicCompactConfig>
|
||||
* each committed summary must be smaller than the content it shadows, and
|
||||
* `compactIfNeeded` may re-compact up to `compactionRetries` extra times before
|
||||
* throwing if the surface still exceeds the threshold.
|
||||
*
|
||||
* @param config - the raw, unresolved backend config.
|
||||
* @returns the validated config with `auto` and `charsPerToken` defaulted.
|
||||
*/
|
||||
export function resolveConfig(config: BasicCompactConfig): ResolvedConfig {
|
||||
const resolved: ResolvedConfig = { auto: true, charsPerToken: 4, ...config }
|
||||
|
||||
@@ -43,7 +43,7 @@ Compaction is serialized via a log-recorded lock: `compactRegion` refuses to sta
|
||||
|
||||
## Events
|
||||
|
||||
The `compact/*` events extend `SessionEventMap` (merge-extensible) via declaration merging — they are session events, not cordis `Events`, and all three are log-only (no `surfaceOp`). Per-event payloads and semantics are in the generated [persistence log event catalog](../../../docs/persistence-catalog/log-events.md).
|
||||
The `compact/*` events extend `SessionEventMap` (merge-extensible) via declaration merging — they are session events, not cordis `Events`, and all three are log-only (no `surfaceOp`). Per-event payloads and semantics are in the generated [persistence log event catalog](../../../docs/persistence-catalog.md).
|
||||
|
||||
## Implementing a backend
|
||||
|
||||
|
||||
@@ -35,11 +35,11 @@ This is the [interface/implementation/consumer seam](../../../docs/rfc/implement
|
||||
|
||||
```ts
|
||||
import type { Config } from '@deepseek-ai/dsh-agent-core'
|
||||
// { agents?, persona? } — the schema is z.intersect([AgentLoop.Config, SystemPrompt.Config]),
|
||||
// { agents?, persona?, toolOrder? } — the schema is z.intersect([AgentLoop.Config, SystemPrompt.Config]),
|
||||
// so validation and defaulting can never drift from the owners'.
|
||||
```
|
||||
|
||||
The bundle FORWARDS each field to the child that owns it: `agents` to `agent-loop` (default `[]`), so each app supplies its own pre-created agents — a stdio app pre-creates a `main`; the ACP app pre-creates none (it creates agents on demand at `session/new`) — and `persona` to `dsh-system-prompt` (default `''`), the deployment's persona section. Forwarding is exactly why the owners can live in the shared spine even though the apps disagree on what to configure.
|
||||
The bundle FORWARDS each field to the child that owns it: `agents` to `agent-loop` (default `[]`), so each app supplies its own pre-created agents — a stdio app pre-creates a `main`; the ACP app pre-creates none (it creates agents on demand at `session/new`) — `persona` to `dsh-system-prompt` (default `''`), the deployment's persona section — and `toolOrder` to `dsh-system-prompt` (absent — lexicographic), the explicit model-facing tool order. Forwarding is exactly why the owners can live in the shared spine even though the apps disagree on what to configure.
|
||||
|
||||
## Why a code bundle, not a shared YAML include
|
||||
|
||||
|
||||
@@ -59,17 +59,20 @@ export const name = 'agent-core'
|
||||
/**
|
||||
* Bundle config: each field forwarded verbatim to the child that owns it —
|
||||
* `agents` to the agent loop (an app that pre-creates no agents, like the ACP
|
||||
* bridge, simply omits it), `persona` to the system-prompt plugin (the
|
||||
* deployment's persona section). Both are optional INPUT here because each
|
||||
* owner's schema supplies the default (`[]` / `''`); the schema is the
|
||||
* INTERSECTION of the owners' own schemas, so validation and defaulting can
|
||||
* never drift from them.
|
||||
* bridge, simply omits it), `persona` and `toolOrder` to the system-prompt
|
||||
* plugin (the deployment's persona section and the explicit model-facing tool
|
||||
* order). Every field is optional INPUT here because each owner's schema
|
||||
* supplies the default (`[]` / `''` / absent — lexicographic); the schema is
|
||||
* the INTERSECTION of the owners' own schemas, so validation and defaulting
|
||||
* can never drift from them.
|
||||
*/
|
||||
export interface Config {
|
||||
/** The agent-loop `agents` list (see dsh-agent-loop's `Config`). */
|
||||
agents?: AgentLoopConfig['agents']
|
||||
/** The deployment persona (see dsh-system-prompt's `Config`). */
|
||||
persona?: SystemPromptConfig['persona']
|
||||
/** The explicit model-facing tool order (see dsh-system-prompt's `Config`). */
|
||||
toolOrder?: SystemPromptConfig['toolOrder']
|
||||
}
|
||||
|
||||
/** Intersect the owners' schemas so validation + defaulting stay identical. */
|
||||
@@ -78,11 +81,11 @@ export const Config = z.intersect([AgentLoop.Config, SystemPrompt.Config]) as un
|
||||
/**
|
||||
* Load the spine. Each `ctx.plugin(...)` mounts one child of the bundle fiber;
|
||||
* `agent-loop` receives the forwarded `agents` list and `system-prompt` the
|
||||
* forwarded `persona`. Load order is irrelevant (cordis pends each fiber on
|
||||
* its `inject` until the services it needs exist), but the listing mirrors the
|
||||
* dependency layering for readability: the LLM vocabulary and core registries
|
||||
* first, then the dev tripwire and the bash tool consumer, then the loop that
|
||||
* drives them.
|
||||
* forwarded `persona` and `toolOrder`. Load order is irrelevant (cordis pends
|
||||
* each fiber on its `inject` until the services it needs exist), but the
|
||||
* listing mirrors the dependency layering for readability: the LLM vocabulary
|
||||
* and core registries first, then the dev tripwire and the bash tool consumer,
|
||||
* then the loop that drives them.
|
||||
*/
|
||||
export function apply(ctx: Context, config: Config): void {
|
||||
ctx.plugin(Timer)
|
||||
@@ -91,8 +94,13 @@ export function apply(ctx: Context, config: Config): void {
|
||||
// The forwarded fields are validated + defaulted by this bundle's intersected
|
||||
// schema before apply runs, so the ?? fallbacks only narrow the
|
||||
// optional-input TYPES — they mirror the owners' schema defaults, never
|
||||
// introduce different ones.
|
||||
ctx.plugin(SystemPrompt, { persona: config.persona ?? '' })
|
||||
// introduce different ones. toolOrder has no owner-supplied default value —
|
||||
// ABSENT means "lexicographic order" — so it is forwarded conditionally
|
||||
// rather than via ??.
|
||||
ctx.plugin(SystemPrompt, {
|
||||
persona: config.persona ?? '',
|
||||
...config.toolOrder !== undefined ? { toolOrder: config.toolOrder } : {},
|
||||
})
|
||||
ctx.plugin(ToolRegistry)
|
||||
ctx.plugin(AgentRegistry)
|
||||
ctx.plugin(invariants)
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import Loader from '@cordisjs/plugin-loader'
|
||||
import { TOOL_ORDER_REST } from '@deepseek-ai/dsh-system-prompt'
|
||||
import * as agentCore from '../src/index.ts'
|
||||
import { AgentId } from '@deepseek-ai/dsh-agent'
|
||||
|
||||
@@ -67,6 +68,23 @@ describe('dsh-agent-core bundle', () => {
|
||||
await ctx.fiber.dispose()
|
||||
})
|
||||
|
||||
it('forwards toolOrder to the system-prompt assembly', async () => {
|
||||
const ctx = await mount({ toolOrder: ['zulu', TOOL_ORDER_REST] })
|
||||
// The bundle's own bash tools pend on the absent `ctx.bash` executor in
|
||||
// this providerless mount, so register two plain tools to order.
|
||||
for (const name of ['alpha', 'zulu']) {
|
||||
ctx.get('tools')!.register({
|
||||
name,
|
||||
description: name,
|
||||
parameters: {},
|
||||
execute: async () => [],
|
||||
})
|
||||
}
|
||||
const assembly = await ctx.get('systemPrompt')!.assemble()
|
||||
expect(assembly.tools.map(tool => tool.name)).toEqual(['zulu', 'alpha'])
|
||||
await ctx.fiber.dispose()
|
||||
})
|
||||
|
||||
it('re-exports the loop config schema as its own', () => {
|
||||
expect(agentCore.Config).toBeDefined()
|
||||
expect(agentCore.name).toBe('agent-core')
|
||||
|
||||
458
packages/core/agent-core/tests/gen-config-catalog.spec.ts
Normal file
458
packages/core/agent-core/tests/gen-config-catalog.spec.ts
Normal file
@@ -0,0 +1,458 @@
|
||||
/**
|
||||
* Negative-path tests for the config catalog generator (`scripts/gen-config-catalog.ts`).
|
||||
*
|
||||
* The generated catalog is frozen by a regenerate-and-diff freshness gate, so
|
||||
* the freshness half is exercised by `pnpm run verify-config-catalog` in CI.
|
||||
* What a freshness diff CANNOT prove is that the generator REJECTS malformed
|
||||
* source the way it promises to — an unclassifiable package, an undocumented
|
||||
* config field, a schema key the config type does not declare, or a referenced
|
||||
* type name that resolves nowhere. These tests drive `collectConfigCatalog()`
|
||||
* against synthetic fixture packages to prove each guard fires (and that
|
||||
* well-formed packages classify and extract correctly), mirroring the
|
||||
* negative tests for gen-cordis-catalog. The spec lives in this package
|
||||
* because agent-core is the config-composition plugin (its schema is the
|
||||
* intersection of its children's), the shape the generator's cross-package
|
||||
* folding exists for.
|
||||
*/
|
||||
|
||||
import { mkdtempSync, mkdirSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { collectConfigCatalog, render } from '../../../../scripts/gen-config-catalog.ts'
|
||||
|
||||
/** Write one fixture package (package.json + src files) under a scan root. */
|
||||
function writePkg(root: string, dir: string, name: string, files: Record<string, string>): void {
|
||||
const pkgDir = join(root, 'packages', dir)
|
||||
mkdirSync(join(pkgDir, 'src'), { recursive: true })
|
||||
writeFileSync(join(pkgDir, 'package.json'), JSON.stringify({ name }))
|
||||
for (const [rel, text] of Object.entries(files)) writeFileSync(join(pkgDir, rel), text)
|
||||
}
|
||||
|
||||
const roots: string[] = []
|
||||
const makeRoot = (): string => {
|
||||
const root = mkdtempSync(join(tmpdir(), 'config-catalog-'))
|
||||
roots.push(root)
|
||||
return root
|
||||
}
|
||||
/** One-package fixture: the common case. */
|
||||
const make = (files: Record<string, string>, name = '@fix/one'): string => {
|
||||
const root = makeRoot()
|
||||
writePkg(root, 'group/one', name, files)
|
||||
return root
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
while (roots.length) rmSync(roots.pop()!, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
const DOCUMENTED_CONFIG = `/** Fixture config. */
|
||||
export interface Config {
|
||||
/** A knob. */
|
||||
knob?: string
|
||||
}
|
||||
`
|
||||
|
||||
describe('gen-config-catalog classification', () => {
|
||||
it('classifies an apply plugin with a config parameter and extracts the paste', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
export const inject = ['tools']
|
||||
${DOCUMENTED_CONFIG}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))
|
||||
expect(entries).toHaveLength(1)
|
||||
expect(entries[0]).toMatchObject({ pkg: '@fix/one', kind: 'config', configTypeName: 'Config', inject: ['tools'] })
|
||||
expect(entries[0]?.pastes?.[0]?.text).toContain('/** A knob. */')
|
||||
})
|
||||
|
||||
it('classifies a default service class, reading its constructor and static inject', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
${DOCUMENTED_CONFIG}
|
||||
/** Fixture service. */
|
||||
export default class Fix {
|
||||
static inject = ['llm']
|
||||
static Config = z.object({ knob: z.string() }) as unknown as z<Config>
|
||||
constructor(ctx: Context, config: Config) {}
|
||||
}
|
||||
`,
|
||||
}))
|
||||
expect(entries[0]).toMatchObject({ kind: 'config', className: 'Fix', inject: ['llm'], schemaKeys: ['knob'] })
|
||||
})
|
||||
|
||||
it('classifies an abstract default class as a seam', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': 'export default abstract class FixSeam { abstract run(): void }\n',
|
||||
}))
|
||||
expect(entries[0]).toMatchObject({ kind: 'seam', className: 'FixSeam' })
|
||||
})
|
||||
|
||||
it('classifies a plugin whose apply takes no config as no-config', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': 'import type { Context } from \'cordis\'\n/** Load. */\nexport function apply(ctx: Context): void {}\n',
|
||||
}))
|
||||
expect(entries[0]?.kind).toBe('no-config')
|
||||
})
|
||||
|
||||
it('classifies a module with neither default export nor apply as a library', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': 'export const helper = 1\n',
|
||||
}))
|
||||
expect(entries[0]?.kind).toBe('library')
|
||||
})
|
||||
|
||||
it('hard-errors on a package with no entry file', () => {
|
||||
const root = makeRoot()
|
||||
mkdirSync(join(root, 'packages', 'group', 'one'), { recursive: true })
|
||||
writeFileSync(join(root, 'packages', 'group', 'one', 'package.json'), JSON.stringify({ name: '@fix/one' }))
|
||||
expect(() => collectConfigCatalog(root)).toThrow(/entry .* is missing or unreadable/)
|
||||
})
|
||||
|
||||
it('hard-errors on a package.json without a name', () => {
|
||||
const root = makeRoot()
|
||||
mkdirSync(join(root, 'packages', 'group', 'one', 'src'), { recursive: true })
|
||||
writeFileSync(join(root, 'packages', 'group', 'one', 'package.json'), '{}')
|
||||
expect(() => collectConfigCatalog(root)).toThrow(/has no "name"/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('gen-config-catalog config extraction guards', () => {
|
||||
it('hard-errors on a config field with no JSDoc prose', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
export interface Config {
|
||||
knob?: string
|
||||
}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).toThrow(/config field 'Config\.knob' .* has no JSDoc prose/)
|
||||
})
|
||||
|
||||
it('hard-errors on an undocumented field nested in a type literal', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** Entries. */
|
||||
entries: {
|
||||
id: string
|
||||
}[]
|
||||
}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).toThrow(/config field 'Config\.entries\.id' .* has no JSDoc prose/)
|
||||
})
|
||||
|
||||
it('pastes a package-local type transitively and records external refs', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import type { Mode } from './types.ts'
|
||||
import type { Remote } from '@fix/dep'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** The mode. */
|
||||
mode?: Mode
|
||||
/** The remote. */
|
||||
remote?: Remote
|
||||
}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
'src/types.ts': '/** Fixture mode. */\nexport type Mode = \'a\' | \'b\'\n',
|
||||
}))
|
||||
expect(entries[0]?.pastes?.map(p => p.source)).toEqual([
|
||||
'packages/group/one/src/index.ts:5',
|
||||
'packages/group/one/src/types.ts:2',
|
||||
])
|
||||
expect(entries[0]?.refs).toEqual([{ alias: 'Remote', imported: 'Remote', specifier: '@fix/dep' }])
|
||||
})
|
||||
|
||||
it('hard-errors on a referenced type name that resolves nowhere', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** The ghost. */
|
||||
ghost?: Ghost
|
||||
}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).toThrow(/references 'Ghost' .* neither declared in the package, imported, nor a known global/)
|
||||
})
|
||||
|
||||
it('hard-errors on a config type imported from another package', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import type { Config } from '@fix/dep'
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).toThrow(/config type 'Config' is imported from '@fix\/dep'/)
|
||||
})
|
||||
|
||||
it('hard-errors when one name resolves to two different declarations across the closure', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import type { A } from './a.ts'
|
||||
import type { B } from './b.ts'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** A. */
|
||||
a?: A
|
||||
/** B. */
|
||||
b?: B
|
||||
}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
'src/a.ts': '/** First Option. */\nexport interface Option {\n /** X. */\n x?: string\n}\n/** A. */\nexport interface A {\n /** O. */\n o?: Option\n}\n',
|
||||
'src/b.ts': '/** Second Option. */\nexport interface Option {\n /** Y. */\n y?: string\n}\n/** B. */\nexport interface B {\n /** O. */\n o?: Option\n}\n',
|
||||
}))).toThrow(/type name 'Option' resolves to two different declarations/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('gen-config-catalog schema cross-check', () => {
|
||||
it('accepts a chained schema whose keys all appear on the config type', () => {
|
||||
const entries = collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
${DOCUMENTED_CONFIG}
|
||||
export const Config: z<Config> = z.object({ knob: z.string() }).default({})
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))
|
||||
expect(entries[0]?.schemaKeys).toEqual(['knob'])
|
||||
})
|
||||
|
||||
it('hard-errors on a schema key the config type does not declare', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
${DOCUMENTED_CONFIG}
|
||||
export const Config: z<Config> = z.object({ knob: z.string(), hidden: z.number() })
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).toThrow(/schema validates key 'hidden' but config type 'Config' declares no such member/)
|
||||
})
|
||||
|
||||
it('hard-errors on a NESTED schema key the config type does not declare', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** Entries. */
|
||||
entries: {
|
||||
/** Id. */
|
||||
id: string
|
||||
}[]
|
||||
}
|
||||
export const Config: z<Config> = z.object({ entries: z.array(z.object({ id: z.string(), ghost: z.string() })) })
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).toThrow(/schema validates key 'entries\[\]\.ghost'/)
|
||||
})
|
||||
|
||||
it('resolves nested keys through a workspace-imported intersection part (re-export chains included)', () => {
|
||||
const root = makeRoot()
|
||||
writePkg(root, 'group/dep', '@fix/dep', {
|
||||
'src/index.ts': 'export * from \'./types.ts\'\n',
|
||||
'src/types.ts': '/** Shared options. */\nexport interface Opts {\n /** Model. */\n model?: string\n}\n',
|
||||
})
|
||||
writePkg(root, 'group/one', '@fix/one', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import type { Opts } from '@fix/dep'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** Entries. */
|
||||
entries: (Opts & {
|
||||
/** Id. */
|
||||
id: string
|
||||
})[]
|
||||
}
|
||||
export const Config: z<Config> = z.object({ entries: z.array(z.object({ id: z.string(), model: z.string() })) })
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
})
|
||||
expect(() => collectConfigCatalog(root)).not.toThrow()
|
||||
})
|
||||
|
||||
it('resolves nested keys through a Partial<> wrapper', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
/** Caps. */
|
||||
export interface Caps {
|
||||
/** X. */
|
||||
x?: boolean
|
||||
}
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** Capabilities. */
|
||||
capabilities?: Partial<Caps>
|
||||
}
|
||||
export const Config: z<Config> = z.object({ capabilities: z.object({ x: z.boolean() }) })
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).not.toThrow()
|
||||
})
|
||||
|
||||
it('leaves a nested key under an external (unresolvable) type unreported', () => {
|
||||
expect(() => collectConfigCatalog(make({
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import type { External } from 'some-external-pkg'
|
||||
/** Fixture config. */
|
||||
export interface Config {
|
||||
/** Options. */
|
||||
options?: External
|
||||
}
|
||||
export const Config: z<Config> = z.object({ options: z.object({ whatever: z.string() }) })
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
}))).not.toThrow()
|
||||
})
|
||||
|
||||
it('folds an intersected workspace schema into the subset check', () => {
|
||||
const root = makeRoot()
|
||||
writePkg(root, 'group/leaf', '@fix/leaf', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
/** Leaf config. */
|
||||
export interface Config {
|
||||
/** Leaf knob. */
|
||||
leaf?: string
|
||||
}
|
||||
/** Leaf service. */
|
||||
export default class Leaf {
|
||||
static Config = z.object({ leaf: z.string() }) as unknown as z<Config>
|
||||
constructor(ctx: Context, config: Config) {}
|
||||
}
|
||||
`,
|
||||
})
|
||||
writePkg(root, 'group/bundle', '@fix/bundle', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import Leaf from '@fix/leaf'
|
||||
/** Bundle config. */
|
||||
export interface Config {
|
||||
/** Forwarded leaf knob. */
|
||||
leaf?: string
|
||||
}
|
||||
export const Config = z.intersect([Leaf.Config]) as unknown as z<Config>
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
})
|
||||
const entries = collectConfigCatalog(root)
|
||||
expect(entries.find(e => e.pkg === '@fix/bundle')?.schemaComposes).toEqual(['@fix/leaf'])
|
||||
})
|
||||
|
||||
it('resolves composed nested keys through an indexed-access forwarder', () => {
|
||||
const root = makeRoot()
|
||||
writePkg(root, 'group/leaf', '@fix/leaf', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
/** Leaf config. */
|
||||
export interface Config {
|
||||
/** Agents. */
|
||||
agents: {
|
||||
/** Id. */
|
||||
id: string
|
||||
}[]
|
||||
}
|
||||
/** Leaf service. */
|
||||
export default class Leaf {
|
||||
static Config = z.object({ agents: z.array(z.object({ id: z.string() })) }) as unknown as z<Config>
|
||||
constructor(ctx: Context, config: Config) {}
|
||||
}
|
||||
`,
|
||||
})
|
||||
writePkg(root, 'group/bundle', '@fix/bundle', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import Leaf, { type Config as LeafConfig } from '@fix/leaf'
|
||||
/** Bundle config forwarding the leaf's agents list. */
|
||||
export interface Config {
|
||||
/** Forwarded agents list. */
|
||||
agents?: LeafConfig['agents']
|
||||
}
|
||||
export const Config = z.intersect([Leaf.Config]) as unknown as z<Config>
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
})
|
||||
expect(() => collectConfigCatalog(root)).not.toThrow()
|
||||
})
|
||||
|
||||
it('hard-errors when an intersected schema key is missing from the bundle config type', () => {
|
||||
const root = makeRoot()
|
||||
writePkg(root, 'group/leaf', '@fix/leaf', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
/** Leaf config. */
|
||||
export interface Config {
|
||||
/** Leaf knob. */
|
||||
leaf?: string
|
||||
}
|
||||
/** Leaf service. */
|
||||
export default class Leaf {
|
||||
static Config = z.object({ leaf: z.string() }) as unknown as z<Config>
|
||||
constructor(ctx: Context, config: Config) {}
|
||||
}
|
||||
`,
|
||||
})
|
||||
writePkg(root, 'group/bundle', '@fix/bundle', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import Leaf from '@fix/leaf'
|
||||
/** Bundle config that forgot to declare the forwarded field. */
|
||||
export interface Config {
|
||||
/** Unrelated. */
|
||||
other?: string
|
||||
}
|
||||
export const Config = z.intersect([Leaf.Config]) as unknown as z<Config>
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
})
|
||||
expect(() => collectConfigCatalog(root)).toThrow(/schema validates key 'leaf' but config type 'Config' declares no such member/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('gen-config-catalog render', () => {
|
||||
it('renders sections, fences, and the terse classification lists', () => {
|
||||
const root = makeRoot()
|
||||
writePkg(root, 'group/one', '@fix/one', {
|
||||
'src/index.ts': `import type { Context } from 'cordis'
|
||||
${DOCUMENTED_CONFIG}
|
||||
/** Load. */
|
||||
export function apply(ctx: Context, config: Config): void {}
|
||||
`,
|
||||
})
|
||||
writePkg(root, 'group/lib', '@fix/lib', { 'src/index.ts': 'export const helper = 1\n' })
|
||||
writePkg(root, 'group/seam', '@fix/seam', {
|
||||
'src/index.ts': 'export default abstract class Seam { abstract run(): void }\n',
|
||||
})
|
||||
const page = render(collectConfigCatalog(root))
|
||||
expect(page).toContain('## `@fix/one`')
|
||||
expect(page).toContain('```ts config-catalog')
|
||||
expect(page).toContain('/** A knob. */')
|
||||
expect(page).toContain('- `@fix/lib` ([`packages/group/lib/src/index.ts`](../packages/group/lib/src/index.ts))')
|
||||
expect(page).toContain('- `@fix/seam` — abstract `Seam`')
|
||||
})
|
||||
})
|
||||
@@ -22,6 +22,10 @@ import { isTurnOpen, lastTurnNumber, runLoop } from './loop.ts'
|
||||
* the agent/* event taxonomy — plugins never need this class.
|
||||
*/
|
||||
export class ReactLoopAgent implements Agent {
|
||||
/**
|
||||
* The queued + steering FIFOs behind {@link send}/{@link steer}. Public so
|
||||
* the driver loop can drain it; {@link cancel} clears it wholesale.
|
||||
*/
|
||||
readonly inbox = new Inbox()
|
||||
|
||||
private _status: AgentStatus = 'idle'
|
||||
@@ -256,6 +260,8 @@ export class ReactLoopAgent implements Agent {
|
||||
* promise (unblocking the idle wait), releases any `whenIdle` waiters, and
|
||||
* aborts the current request if any. The returned `agent.done` promise
|
||||
* resolves once the loop exits.
|
||||
* @returns the disposer — idempotent and infallible (it runs inside the
|
||||
* fiber's LIFO disposal chain, where a throw would skip later disposers).
|
||||
*/
|
||||
start(): () => void {
|
||||
this.done = runLoop(this.ctx, this, {
|
||||
|
||||
@@ -24,30 +24,47 @@ export class Inbox {
|
||||
private steeringMessages: InboxMessage[] = []
|
||||
private wakeup: (() => void) | undefined
|
||||
|
||||
/** Resolves when a queued message arrives (used by the idle loop). */
|
||||
/** True while queued messages are pending — read by the idle wait's fast path and the loop's turn-start checks. */
|
||||
get hasQueued(): boolean {
|
||||
return this.queuedMessages.length > 0
|
||||
}
|
||||
|
||||
/** True while steering messages are pending — read by `cancel()`'s arm gate and the loop's stop-override check. */
|
||||
get hasSteering(): boolean {
|
||||
return this.steeringMessages.length > 0
|
||||
}
|
||||
|
||||
/**
|
||||
* Add a message to the queued FIFO and wake a parked {@link waitForQueued}.
|
||||
* @param message - the message to queue for the next turn start.
|
||||
*/
|
||||
enqueue(message: InboxMessage): void {
|
||||
this.queuedMessages.push(message)
|
||||
this.wakeup?.()
|
||||
}
|
||||
|
||||
/**
|
||||
* Add a message to the steering FIFO. Deliberately no wakeup: steering is
|
||||
* drained between steps of a running turn, never by the idle wait —
|
||||
* `Agent.steer()` on an idle agent falls back to `send()` instead.
|
||||
* @param message - the message to inject between steps of the running turn.
|
||||
*/
|
||||
steer(message: InboxMessage): void {
|
||||
this.steeringMessages.push(message)
|
||||
}
|
||||
|
||||
/** Drain all queued messages (turn start). */
|
||||
/**
|
||||
* Drain all queued messages (turn start).
|
||||
* @returns the drained messages in arrival order; the queued FIFO is left empty.
|
||||
*/
|
||||
drainQueued(): InboxMessage[] {
|
||||
return this.queuedMessages.splice(0)
|
||||
}
|
||||
|
||||
/** Drain all steering messages (between steps). */
|
||||
/**
|
||||
* Drain all steering messages (between steps).
|
||||
* @returns the drained messages in arrival order; the steering FIFO is left empty.
|
||||
*/
|
||||
drainSteering(): InboxMessage[] {
|
||||
return this.steeringMessages.splice(0)
|
||||
}
|
||||
@@ -62,7 +79,12 @@ export class Inbox {
|
||||
this.steeringMessages.length = 0
|
||||
}
|
||||
|
||||
/** Wait until a queued message arrives or `cancel` resolves. */
|
||||
/**
|
||||
* Wait until a queued message arrives or `cancel` resolves.
|
||||
* @param cancel - a promise whose resolution abandons the wait without a
|
||||
* message (the driver loop passes the agent's disposed promise so a parked
|
||||
* loop can exit).
|
||||
*/
|
||||
waitForQueued(cancel: Promise<void>): Promise<void> {
|
||||
if (this.hasQueued) return Promise.resolve()
|
||||
const { promise, resolve } = Promise.withResolvers<void>()
|
||||
|
||||
@@ -29,9 +29,14 @@ declare module 'cordis' {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Plugin config: the agents to create — or resume, via `resumeSessionId` —
|
||||
* declaratively at startup, so a cordis.yml deployment needs no code.
|
||||
*/
|
||||
export interface Config {
|
||||
/** Agents created from configuration at startup. */
|
||||
agents: (AgentOptions & {
|
||||
/** Agent id to register under; also seeds the fresh per-run session id (`${id}-session-<uuid>`). */
|
||||
id: AgentId
|
||||
/**
|
||||
* If set, the config agent RESUMES this persisted session id instead of
|
||||
|
||||
@@ -185,6 +185,9 @@ export interface LoopHandle {
|
||||
* re-enqueue leftover steering as queued ⟵ steering is never stranded
|
||||
* idle (emit agent/status) unless more queued
|
||||
* ```
|
||||
* @param ctx - the plugin context the loop reaches events (agent/…, session/flush) and services (systemPrompt, llm, tools) through.
|
||||
* @param agent - the agent this invocation drives for its whole lifetime (its inbox, session, and options).
|
||||
* @param handle - the bridge to the agent's mutable state: status/abort setters plus the disposal and cancel-marker reads.
|
||||
*/
|
||||
export async function runLoop(ctx: Context, agent: ReactLoopAgent, handle: LoopHandle): Promise<void> {
|
||||
// Per-instance transmission bookkeeping: whether THIS loop instance has
|
||||
@@ -875,7 +878,11 @@ function withoutToolCalls(message: Message): Message {
|
||||
return { ...message, content: message.content.filter(block => block.type !== 'tool-call') }
|
||||
}
|
||||
|
||||
/** The last turn number in a (possibly seeded) session log, or 0. */
|
||||
/**
|
||||
* The last turn number in a (possibly seeded) session log, or 0.
|
||||
* @param session - the session whose log is scanned for the latest `turn/start`.
|
||||
* @returns the latest `turn/start`'s turn number, or 0 when the log has none (the next turn is this plus one).
|
||||
*/
|
||||
export function lastTurnNumber(session: Session): number {
|
||||
const lastStart = session.events.findLast(event => event.type === 'turn/start')
|
||||
return lastStart?.data.turn ?? 0
|
||||
@@ -889,6 +896,8 @@ export function lastTurnNumber(session: Session): number {
|
||||
* returns to idle), so status is not a reliable open-turn signal. Used by
|
||||
* `inject()` to choose between appending into an open turn vs. wrapping the
|
||||
* injection in its own one-shot turn (the turn-enclosure RFC).
|
||||
* @param session - the session whose log is inspected.
|
||||
* @returns true when the log's last turn boundary is a `turn/start` with no matching `turn/end` yet.
|
||||
*/
|
||||
export function isTurnOpen(session: Session): boolean {
|
||||
const last = session.events.findLast(e => e.type === 'turn/start' || e.type === 'turn/end')
|
||||
|
||||
@@ -19,7 +19,10 @@ export interface TransmissionLog {
|
||||
loggedHeader: boolean
|
||||
}
|
||||
|
||||
/** Fresh bookkeeping for a newly-started loop instance. */
|
||||
/**
|
||||
* Fresh bookkeeping for a newly-started loop instance.
|
||||
* @returns state with `loggedHeader` false, so the instance's first request appends an anchoring snapshot.
|
||||
*/
|
||||
export function createTransmissionLog(): TransmissionLog {
|
||||
return { loggedHeader: false }
|
||||
}
|
||||
|
||||
117
packages/core/agent-loop/tests/tool-order.spec.ts
Normal file
117
packages/core/agent-loop/tests/tool-order.spec.ts
Normal file
@@ -0,0 +1,117 @@
|
||||
/**
|
||||
* Loop-level tool-order determinism: the request/header event — and therefore
|
||||
* the frozen request the adapter receives — carries the assembly's canonical
|
||||
* tool order (system-prompt's `toolOrder` config, or lexicographic name
|
||||
* order), regardless of the order tool plugins happened to register in.
|
||||
* Registration order is a plugin-load artifact (concurrent dynamic imports
|
||||
* race), so nothing downstream of the registry may depend on it.
|
||||
*/
|
||||
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService from '@deepseek-ai/dsh-llm'
|
||||
import SessionStore, { foldRequestHeader } from '@deepseek-ai/dsh-session'
|
||||
import SystemPrompt, { TOOL_ORDER_REST } from '@deepseek-ai/dsh-system-prompt'
|
||||
import type { Config as SystemPromptConfig } from '@deepseek-ai/dsh-system-prompt'
|
||||
import ToolRegistry, { defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
|
||||
import AgentLoop, { ReactLoopAgent } from '@deepseek-ai/dsh-agent-loop'
|
||||
import { MockAdapter, textResponse } from './mock-adapter.ts'
|
||||
|
||||
async function harness(adapter: MockAdapter, toolOrder?: SystemPromptConfig['toolOrder']) {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(SystemPrompt, { persona: 'stable base', ...toolOrder !== undefined ? { toolOrder } : {} })
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await ctx.plugin(AgentRegistry)
|
||||
await ctx.plugin(AgentLoop, { agents: [] })
|
||||
ctx.llm.registerAdapter(['mock'], adapter)
|
||||
return ctx
|
||||
}
|
||||
|
||||
function waitForIdle(ctx: Context, agent: ReactLoopAgent): Promise<void> {
|
||||
return new Promise((resolve) => {
|
||||
const dispose = ctx.on('agent/status', (subject, status) => {
|
||||
if (subject === agent && status === 'idle') {
|
||||
dispose()
|
||||
resolve()
|
||||
}
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
function registerNamed(ctx: Context, name: string) {
|
||||
ctx.tools.register(defineTool({
|
||||
name,
|
||||
description: `the ${name} tool`,
|
||||
parameters: {},
|
||||
async execute() {
|
||||
return [{ type: 'text', text: name }]
|
||||
},
|
||||
}))
|
||||
}
|
||||
|
||||
/** Run one text-only turn and return the harness context + agent. */
|
||||
async function runTurn(registrationOrder: string[], toolOrder?: SystemPromptConfig['toolOrder']) {
|
||||
const adapter = new MockAdapter([textResponse('done')])
|
||||
const ctx = await harness(adapter, toolOrder)
|
||||
for (const name of registrationOrder) registerNamed(ctx, name)
|
||||
const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
|
||||
agent.send([{ type: 'text', text: 'go' }])
|
||||
await waitForIdle(ctx, agent)
|
||||
return { ctx, agent, adapter }
|
||||
}
|
||||
|
||||
describe('loop-level canonical tool order', () => {
|
||||
it('logs the request/header with tools in canonical order, not registration order', async () => {
|
||||
const { agent, adapter } = await runTurn(['zulu', 'alpha', 'mike'])
|
||||
const header = foldRequestHeader(agent.session.events)
|
||||
expect(header?.tools?.map(tool => tool.name)).toEqual(['alpha', 'mike', 'zulu'])
|
||||
// The dispatched request is built FROM the logged header (whose tools the
|
||||
// assembly already canonicalized) and reaches the adapter deep-frozen —
|
||||
// the marker the reconstruction invariant keys on.
|
||||
expect(adapter.requests[0]?.tools?.map(tool => tool.name)).toEqual(['alpha', 'mike', 'zulu'])
|
||||
expect(Object.isFrozen(adapter.requests[0])).toBe(true)
|
||||
expect(adapter.requests[0]?.sessionId).toBe(agent.session.id)
|
||||
})
|
||||
|
||||
it('produces the same header order for any registration order', async () => {
|
||||
const first = await runTurn(['alpha', 'mike', 'zulu'])
|
||||
const second = await runTurn(['zulu', 'mike', 'alpha'])
|
||||
const names = (run: typeof first) => foldRequestHeader(run.agent.session.events)?.tools?.map(tool => tool.name)
|
||||
expect(names(first)).toEqual(['alpha', 'mike', 'zulu'])
|
||||
expect(names(second)).toEqual(names(first))
|
||||
})
|
||||
|
||||
it('honors a configured toolOrder in the logged header and the dispatched request', async () => {
|
||||
const { agent, adapter } = await runTurn(['alpha', 'zulu', 'mike'], ['zulu', TOOL_ORDER_REST])
|
||||
const header = foldRequestHeader(agent.session.events)
|
||||
expect(header?.tools?.map(tool => tool.name)).toEqual(['zulu', 'alpha', 'mike'])
|
||||
expect(adapter.requests[0]?.tools?.map(tool => tool.name)).toEqual(['zulu', 'alpha', 'mike'])
|
||||
expect(Object.isFrozen(adapter.requests[0])).toBe(true)
|
||||
})
|
||||
|
||||
it('fails the turn — no model request — when toolOrder names an unregistered tool', async () => {
|
||||
// The assemble rejection escapes to runTurn's outer catch: the open turn
|
||||
// closes with an `error` reason (agent/error mirrors it), no step opens,
|
||||
// no request/header is logged, the adapter never sees a request, and the
|
||||
// agent returns to idle — a misconfigured deployment fails every turn
|
||||
// deterministically instead of silently reordering nothing.
|
||||
const adapter = new MockAdapter([textResponse('never sent')])
|
||||
const ctx = await harness(adapter, ['ghost', TOOL_ORDER_REST])
|
||||
registerNamed(ctx, 'alpha')
|
||||
const errors: Error[] = []
|
||||
ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error))
|
||||
const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
|
||||
agent.send([{ type: 'text', text: 'go' }])
|
||||
await waitForIdle(ctx, agent)
|
||||
expect(adapter.requests).toHaveLength(0)
|
||||
expect(errors.map(e => e.message)).toEqual(['toolOrder lists unregistered tool "ghost"; registered tools: alpha'])
|
||||
expect(foldRequestHeader(agent.session.events)).toBeUndefined()
|
||||
const end = agent.session.events.find(e => e.type === 'turn/end')
|
||||
expect(end?.type === 'turn/end' && end.data.reason).toMatchObject({ kind: 'error', step: 1 })
|
||||
// The turn is balanced (turn/start → turn/end) with no step events inside.
|
||||
expect(agent.session.events.some(e => e.type === 'step/start')).toBe(false)
|
||||
})
|
||||
})
|
||||
@@ -50,7 +50,11 @@ import type {} from '@deepseek-ai/dsh-system-prompt'
|
||||
/** Identifies one live agent in the registry. */
|
||||
export type AgentId = Branded<'AgentId'>
|
||||
|
||||
/** Brand a string as an {@link AgentId}. */
|
||||
/**
|
||||
* Brand a string as an {@link AgentId}.
|
||||
* @param id - the raw agent id string.
|
||||
* @returns the same string, branded (a compile-time cast — no runtime cost).
|
||||
*/
|
||||
export function AgentId(id: string): AgentId {
|
||||
return id as AgentId
|
||||
}
|
||||
@@ -80,10 +84,22 @@ export interface AgentOptions {
|
||||
model?: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for {@link Agent.send}/{@link Agent.steer}/{@link Agent.inject}. An
|
||||
* absent `source` resolves to `{ kind: 'user' }`, so a plugin supplying content
|
||||
* must label itself here or its message is recorded as a user prompt (see
|
||||
* {@link HookContext} on why that label is load-bearing).
|
||||
*/
|
||||
export interface SendOptions {
|
||||
source?: MessageSource
|
||||
}
|
||||
|
||||
/**
|
||||
* An agent's lifecycle state, emitted on every transition as `agent/status`:
|
||||
* `idle` (parked, waiting for queued work), `running` (a turn is in progress),
|
||||
* `disposed` (terminal — no transition leaves it, and `send`/`steer`/`inject`
|
||||
* throw).
|
||||
*/
|
||||
export type AgentStatus = 'idle' | 'running' | 'disposed'
|
||||
|
||||
/**
|
||||
|
||||
555
packages/core/agent/tests/verify-export-jsdoc.spec.ts
Normal file
555
packages/core/agent/tests/verify-export-jsdoc.spec.ts
Normal file
@@ -0,0 +1,555 @@
|
||||
/**
|
||||
* Negative-path tests for the export-surface JSDoc gate
|
||||
* (`scripts/verify-export-jsdoc.ts`).
|
||||
*
|
||||
* The gate's positive half runs against the real tree in CI (`pnpm run
|
||||
* verify-export-jsdoc`, part of doc-sync). What that run cannot prove is that
|
||||
* the walk REJECTS an undocumented surface the way it promises to — and that
|
||||
* every deliberate exemption (heritage members, plugin-protocol slots,
|
||||
* constructors, overload implementations, augmentation bodies, re-exports)
|
||||
* actually holds. These tests drive `collectExportJsdocViolations()` against
|
||||
* synthetic fixture packages, mirroring the gen-cordis-catalog negative
|
||||
* tests.
|
||||
*/
|
||||
|
||||
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { dirname, join } from 'node:path'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { collectExportJsdocViolations } from '../../../../scripts/verify-export-jsdoc.ts'
|
||||
|
||||
const roots: string[] = []
|
||||
|
||||
afterEach(() => {
|
||||
while (roots.length) rmSync(roots.pop()!, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
/** Write fixture files under `packages/group/fix/src/` and return the scan root. */
|
||||
function fixture(files: Record<string, string>): string {
|
||||
const root = mkdtempSync(join(tmpdir(), 'export-jsdoc-'))
|
||||
roots.push(root)
|
||||
for (const [rel, content] of Object.entries(files)) {
|
||||
const abs = join(root, 'packages', 'group', 'fix', 'src', rel)
|
||||
mkdirSync(dirname(abs), { recursive: true })
|
||||
writeFileSync(abs, content)
|
||||
}
|
||||
return root
|
||||
}
|
||||
|
||||
/** Single-file fixture shorthand: the content becomes `src/index.ts`. */
|
||||
const make = (content: string): string => fixture({ 'index.ts': content })
|
||||
|
||||
describe('verify-export-jsdoc functions and consts', () => {
|
||||
it('accepts a fully documented surface', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/**
|
||||
* Add one to a count.
|
||||
* @param n - the count to bump.
|
||||
* @returns the count plus one.
|
||||
*/
|
||||
export function bump(n: number): number { return n + 1 }
|
||||
|
||||
/**
|
||||
* Fire-and-forget (void needs no @returns).
|
||||
* @param flag - whether to arm.
|
||||
*/
|
||||
export function poke(flag: boolean): void { void flag }
|
||||
|
||||
/** The default retry budget. */
|
||||
export const RETRIES = 3
|
||||
|
||||
/**
|
||||
* Halve a count.
|
||||
* @param n - the count to halve.
|
||||
* @returns the count halved.
|
||||
*/
|
||||
export const halve = (n: number): number => n / 2
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('flags an exported function with no JSDoc at all', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'export function bare(): void {}\n',
|
||||
))).toEqual([expect.stringMatching(/exported function 'bare' .* has no JSDoc\./)])
|
||||
})
|
||||
|
||||
it('flags a missing @param and a missing @returns', () => {
|
||||
const violations = collectExportJsdocViolations(make(
|
||||
'/** Docs without tags. */\nexport function f(x: number): number { return x }\n',
|
||||
))
|
||||
expect(violations).toEqual([
|
||||
expect.stringMatching(/exported function 'f' .* is missing @param x\./),
|
||||
expect.stringMatching(/exported function 'f' .* is missing @returns \(return type: number\)\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('flags an unannotated (inferred) return type', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/**\n * Docs.\n * @param x - value.\n */\nexport function f(x: number) { return x }\n',
|
||||
))).toEqual([expect.stringMatching(/no return type annotation/)])
|
||||
})
|
||||
|
||||
it('flags tags-only JSDoc with no description prose', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/**\n * @param x - value.\n */\nexport function f(x: number): void {}\n',
|
||||
))).toEqual([expect.stringMatching(/no description prose above its block tags/)])
|
||||
})
|
||||
|
||||
it('flags a stale @param and a binding-pattern parameter', () => {
|
||||
const violations = collectExportJsdocViolations(make(
|
||||
'/**\n * Docs.\n * @param ghost - not real.\n */\nexport function f({ a }: { a: number }): void {}\n',
|
||||
))
|
||||
expect(violations).toEqual([
|
||||
expect.stringMatching(/parameter '\{ a \}' is a binding pattern; the export surface needs simple identifier parameters/),
|
||||
expect.stringMatching(/@param ghost does not match any parameter \(stale tag\?\)/),
|
||||
])
|
||||
})
|
||||
|
||||
it('exempts a `this` receiver annotation from @param', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/**\n * Docs.\n * @param x - value.\n */\nexport function f(this: object, x: number): void {}\n',
|
||||
))).toEqual([])
|
||||
})
|
||||
|
||||
it('waives @returns for a declarator-annotated const but not an unannotated one', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
type Fn = (x: number) => number
|
||||
/**
|
||||
* Uses the named signature.
|
||||
* @param x - value.
|
||||
*/
|
||||
export const good: Fn = x => x
|
||||
/**
|
||||
* No signature anywhere.
|
||||
* @param x - value.
|
||||
*/
|
||||
export const bad = (x: number) => x
|
||||
`))).toEqual([expect.stringMatching(/exported const 'bad' .* has no return type annotation/)])
|
||||
})
|
||||
|
||||
it('requires description prose on a non-function const', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'export const LIMIT = 10\n',
|
||||
))).toEqual([expect.stringMatching(/exported const 'LIMIT' .* has no JSDoc\./)])
|
||||
})
|
||||
})
|
||||
|
||||
describe('verify-export-jsdoc type-level exports', () => {
|
||||
it('requires description prose on interfaces, type aliases, and enums', () => {
|
||||
const violations = collectExportJsdocViolations(make(
|
||||
'export interface I { a: number }\nexport type T = number\nexport enum E { A }\n',
|
||||
))
|
||||
expect(violations).toEqual([
|
||||
expect.stringMatching(/exported interface 'I' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported type 'T' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported enum 'E' .* has no JSDoc\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('skips `declare module` augmentation bodies (the cordis gate owns them)', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
"declare module 'cordis' {\n interface Events {\n 'fix/x'(): void\n }\n}\nexport {}\n",
|
||||
))).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('verify-export-jsdoc export forms', () => {
|
||||
it('resolves an `export { … }` list to the local declaration', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'function f(): void {}\nexport { f }\n',
|
||||
))).toEqual([expect.stringMatching(/exported function 'f' .* has no JSDoc\./)])
|
||||
})
|
||||
|
||||
it('does not treat a never-exported sibling declarator as surface (review round 2)', () => {
|
||||
// `export { publicValue }` resolves to the whole variable statement; only
|
||||
// the named declarator is surface — the gate must not demand JSDoc for
|
||||
// the private sibling sharing the statement.
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** The public knob. */\nconst publicValue = 1, privateHelper = 2\nexport { publicValue }\nvoid privateHelper\n',
|
||||
))).toEqual([])
|
||||
})
|
||||
|
||||
it('unions declarators across multiple export lists over one statement (review round 2)', () => {
|
||||
// Two lists each name one declarator of the same undocumented statement:
|
||||
// both are surface (deduplicating on first resolution would drop `b`),
|
||||
// while the never-exported `c` stays out.
|
||||
const violations = collectExportJsdocViolations(make(
|
||||
'const a = 1, b = 2, c = 3\nexport { a }\nexport { b }\nvoid c\n',
|
||||
))
|
||||
expect(violations).toEqual([
|
||||
expect.stringMatching(/exported const 'a' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported const 'b' .* has no JSDoc\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('scopes a default-export identifier to its own declarator (review round 2)', () => {
|
||||
// `export default` of an identifier reaches the statement through the
|
||||
// same name lookup as an export list; the sibling stays private.
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** The app entry. */\nconst app = 1, scratch = 2\nexport default app\nvoid scratch\n',
|
||||
))).toEqual([])
|
||||
})
|
||||
|
||||
it('reports a re-exported module once, at its defining file', () => {
|
||||
const violations = collectExportJsdocViolations(fixture({
|
||||
'index.ts': "export * from './other.ts'\n",
|
||||
'other.ts': 'export function f(): void {}\n',
|
||||
}))
|
||||
expect(violations).toEqual([expect.stringMatching(/other\.ts:1\) has no JSDoc\./)])
|
||||
})
|
||||
|
||||
it('exempts overload implementations when the signatures are documented', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/**
|
||||
* From a number.
|
||||
* @param x - the number.
|
||||
* @returns its text.
|
||||
*/
|
||||
export function f(x: number): string
|
||||
/**
|
||||
* From a flag.
|
||||
* @param x - the flag.
|
||||
* @returns its text.
|
||||
*/
|
||||
export function f(x: boolean): string
|
||||
export function f(x: number | boolean): string { return String(x) }
|
||||
`))).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('verify-export-jsdoc classes', () => {
|
||||
it('flags an undocumented class, method, property, and accessor', () => {
|
||||
const violations = collectExportJsdocViolations(make(`
|
||||
export class C {
|
||||
state = 1
|
||||
get view(): number { return this.state }
|
||||
run(x: number): number { return x }
|
||||
}
|
||||
`))
|
||||
expect(violations).toEqual([
|
||||
expect.stringMatching(/exported class 'C' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported class property 'C.state' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported class accessor 'C.view' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported class method 'C.run' .* has no JSDoc\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('exempts members declared by an extends/implements heritage type', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/**
|
||||
* Do it.
|
||||
* @param x - input.
|
||||
* @returns output.
|
||||
*/
|
||||
abstract run(x: number): number
|
||||
}
|
||||
/** Iface. */
|
||||
export interface Sized {
|
||||
/** Byte size. */
|
||||
size: number
|
||||
}
|
||||
/** Impl. */
|
||||
export class Impl extends Base implements Sized {
|
||||
size = 0
|
||||
run(x: number): number { return x }
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('skips private/protected/#private members and constructors', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Documented. */
|
||||
export class C {
|
||||
#secret = 1
|
||||
private hidden(): void {}
|
||||
protected hook(): void {}
|
||||
constructor(x: number) { void x }
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('exempts plugin-protocol statics but checks other statics', () => {
|
||||
const violations = collectExportJsdocViolations(make(`
|
||||
/** Plugin. */
|
||||
export class C {
|
||||
static Config = { a: 1 }
|
||||
static inject = ['bash']
|
||||
static reusable = true
|
||||
static other = 1
|
||||
}
|
||||
`))
|
||||
expect(violations).toEqual([expect.stringMatching(/exported class property 'C.other' .* has no JSDoc\./)])
|
||||
})
|
||||
|
||||
it("covers a set accessor by the getter's doc", () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Documented. */
|
||||
export class C {
|
||||
/** The current width. */
|
||||
get width(): number { return 1 }
|
||||
set width(_v: number) {}
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('verify-export-jsdoc plugin protocol and namespaces', () => {
|
||||
it('exempts top-level plugin-protocol exports', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
export const name = 'fix'
|
||||
export const inject = ['bash']
|
||||
export const reusable = true
|
||||
export const Config = { parse: true }
|
||||
export function apply(): void {}
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('recurses into namespaces with qualified names and honors the merge idiom', () => {
|
||||
const violations = collectExportJsdocViolations(make(`
|
||||
/** The plugin class. */
|
||||
export class Fix {}
|
||||
export namespace Fix {
|
||||
export interface Config { a: number }
|
||||
}
|
||||
export namespace Loose {
|
||||
export const x = 1
|
||||
}
|
||||
`))
|
||||
expect(violations).toEqual([
|
||||
expect.stringMatching(/exported interface 'Fix.Config' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported namespace 'Loose' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported const 'Loose.x' .* has no JSDoc\./),
|
||||
])
|
||||
})
|
||||
})
|
||||
|
||||
describe('verify-export-jsdoc fail-closed forms (review round 1)', () => {
|
||||
it('checks the function contract on a non-identifier default export', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** Doubles. */\nexport default (x: number): number => x * 2\n',
|
||||
))).toEqual([
|
||||
expect.stringMatching(/default export .* is missing @param x\./),
|
||||
expect.stringMatching(/default export .* is missing @returns \(return type: number\)\./),
|
||||
])
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/**\n * Doubles.\n * @param x - the input.\n * @returns twice the input.\n */\nexport default (x: number): number => x * 2\n',
|
||||
))).toEqual([])
|
||||
})
|
||||
|
||||
it('treats an inline function-type annotation as the surface signature', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** Maps a number. */\nexport declare const f: (x: number) => number\n',
|
||||
))).toEqual([
|
||||
expect.stringMatching(/exported const 'f' .* is missing @param x\./),
|
||||
expect.stringMatching(/exported const 'f' .* is missing @returns \(return type: number\)\./),
|
||||
])
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/**\n * Maps a number.\n * @param x - the input.\n * @returns the mapped value.\n */\nexport const f: (x: number) => number = v => v\n',
|
||||
))).toEqual([])
|
||||
})
|
||||
|
||||
it('recurses into an ambient declare namespace where members export implicitly', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'export declare namespace N {\n function f(x: number): number\n}\n',
|
||||
))).toEqual([
|
||||
expect.stringMatching(/exported namespace 'N' .* has no JSDoc\./),
|
||||
expect.stringMatching(/exported function 'N.f' .* has no JSDoc\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('requires an export-import alias to document itself (its target may be unwalked)', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** Holder. */\nexport namespace N {\n /** The value. */\n export const x = 1\n}\nexport import y = N.x\n',
|
||||
))).toEqual([expect.stringMatching(/exported alias 'y' .* has no JSDoc\./)])
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'namespace N {\n export const x = 1\n}\n/** Alias surfacing the internal counter. */\nexport import y = N.x\n',
|
||||
))).toEqual([])
|
||||
})
|
||||
|
||||
it('refuses an export-import alias to a callable, class, or namespace target', () => {
|
||||
const refusal = /exported alias 'g' .* aliases a callable, class, or namespace target/
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'namespace N {\n export function f(x: number): number { return x }\n}\n/** Alias. */\nexport import g = N.f\n',
|
||||
))).toEqual([expect.stringMatching(refusal)])
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'namespace N {\n export class C {\n run(x: number): number { return x }\n }\n}\n/** Alias. */\nexport import g = N.C\n',
|
||||
))).toEqual([expect.stringMatching(refusal)])
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'namespace N {\n export namespace Sub {\n export function f(x: number): number { return x }\n }\n}\n/** Alias. */\nexport import g = N.Sub\n',
|
||||
))).toEqual([expect.stringMatching(refusal)])
|
||||
})
|
||||
|
||||
it('classifies wrapped function initializers and default exports (parens, satisfies)', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'type Fn = (x: number) => number\n/** Wrapped. */\nexport const f = (((x: number): number => x)) satisfies Fn\n',
|
||||
))).toEqual([
|
||||
expect.stringMatching(/exported const 'f' .* is missing @param x\./),
|
||||
expect.stringMatching(/exported const 'f' .* is missing @returns \(return type: number\)\./),
|
||||
])
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'type Fn = (x: number) => number\n/** Wrapped. */\nexport default (((x: number): number => x * 2) satisfies Fn)\n',
|
||||
))).toEqual([
|
||||
expect.stringMatching(/default export .* is missing @param x\./),
|
||||
expect.stringMatching(/default export .* is missing @returns \(return type: number\)\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('treats a single-call-signature type literal as the surface signature', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** Maps. */\nexport declare const f: { (x: number): number }\n',
|
||||
))).toEqual([
|
||||
expect.stringMatching(/exported const 'f' .* is missing @param x\./),
|
||||
expect.stringMatching(/exported const 'f' .* is missing @returns \(return type: number\)\./),
|
||||
])
|
||||
})
|
||||
|
||||
it('refuses a hybrid callable type literal instead of narrowing the check', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'/** Hybrid. */\nexport declare const f: { (x: number): number; flush: () => void }\n',
|
||||
))).toEqual([expect.stringMatching(/exported const 'f'.*callable type literal is not gate-classifiable; extract a named type/)])
|
||||
})
|
||||
|
||||
it('refuses an export-equals assignment instead of failing open', () => {
|
||||
expect(collectExportJsdocViolations(make(
|
||||
'const x = 1\nexport = x\n',
|
||||
))).toEqual([expect.stringMatching(/export-equals assignment .* is not a gate-supported export form/)])
|
||||
})
|
||||
})
|
||||
|
||||
describe('verify-export-jsdoc heritage refinement (review round 1)', () => {
|
||||
it('requires @param for parameters the base member never names', () => {
|
||||
const violations = collectExportJsdocViolations(make(`
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/**
|
||||
* Do it.
|
||||
* @param x - input.
|
||||
* @returns output.
|
||||
*/
|
||||
abstract run(x: number): number
|
||||
}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
override run(x: number, verbose?: boolean): number { return verbose ? x : -x }
|
||||
}
|
||||
`))
|
||||
expect(violations).toEqual([expect.stringMatching(/exported class method 'Impl.run' .* is missing @param verbose\./)])
|
||||
})
|
||||
|
||||
it('does not exempt a public override of a protected-only base member', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/** Subclass hook. */
|
||||
protected hook(): void {}
|
||||
}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
override hook(): void {}
|
||||
}
|
||||
`))).toEqual([expect.stringMatching(/exported class method 'Impl.hook' .* has no JSDoc\./)])
|
||||
})
|
||||
|
||||
it('treats an underscore-prefixed rename of a base parameter as the same parameter', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/**
|
||||
* Load it.
|
||||
* @param cwd - the working directory to scope the lookup.
|
||||
* @returns the loaded value.
|
||||
*/
|
||||
abstract load(cwd: string): number
|
||||
}
|
||||
/** Impl (ignores cwd). */
|
||||
export class Impl extends Base {
|
||||
load(_cwd: string): number { return 1 }
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('flags a binding-pattern parameter an override adds beyond the base', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/**
|
||||
* Do it.
|
||||
* @param x - input.
|
||||
* @returns output.
|
||||
*/
|
||||
abstract run(x: number): number
|
||||
}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
override run(x: number, { verbose }: { verbose?: boolean } = {}): number { return verbose ? x : -x }
|
||||
}
|
||||
`))).toEqual([expect.stringMatching(/exported class method 'Impl.run' .* is a binding pattern/)])
|
||||
})
|
||||
|
||||
it('revives the @returns duty when an override grows a concrete result over a void base', () => {
|
||||
const voidBase = `
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/** Do it (fire-and-forget). */
|
||||
abstract run(): void
|
||||
}
|
||||
`
|
||||
expect(collectExportJsdocViolations(make(`${voidBase}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
override run(): number { return 1 }
|
||||
}
|
||||
`))).toEqual([expect.stringMatching(/exported class method 'Impl.run' .* is missing @returns \(return type: number\)\./)])
|
||||
expect(collectExportJsdocViolations(make(`${voidBase}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
/**
|
||||
* Do it and count.
|
||||
* @returns how many were done.
|
||||
*/
|
||||
override run(): number { return 1 }
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('classifies an unannotated override return over a void base via the checker', () => {
|
||||
const voidBase = `
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/** Do it (fire-and-forget). */
|
||||
abstract run(): void
|
||||
}
|
||||
`
|
||||
expect(collectExportJsdocViolations(make(`${voidBase}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
override run() { return 1 }
|
||||
}
|
||||
`))).toEqual([expect.stringMatching(/exported class method 'Impl.run' .* non-void result its heritage declaration does not document/)])
|
||||
expect(collectExportJsdocViolations(make(`${voidBase}
|
||||
/** Impl (faithful void, no annotation needed). */
|
||||
export class Impl extends Base {
|
||||
override run() {}
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
|
||||
it('keeps the full exemption when the base return already carries the @returns duty', () => {
|
||||
expect(collectExportJsdocViolations(make(`
|
||||
/** Seam. */
|
||||
export abstract class Base {
|
||||
/**
|
||||
* Count things.
|
||||
* @returns the count.
|
||||
*/
|
||||
abstract run(): number
|
||||
}
|
||||
/** Impl. */
|
||||
export class Impl extends Base {
|
||||
override run(): number { return 1 }
|
||||
}
|
||||
`))).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -9,6 +9,7 @@ Creates and holds event-sourced `Session` instances. Persistence is intentionall
|
||||
### Public API
|
||||
|
||||
- `ctx.sessions.create(id?: SessionId, options?: { seed?: SessionEvent[]; meta?: { cwd?: string; parentSession?: SessionId; createdAt?: number; seedLength?: number } }): Session` — Create a session. `options.seed` replays/forks an existing event log; `options.meta` attaches creation metadata (validated absolute `cwd`, `parentSession` lineage, seed boundary) as the immutable `SessionHeader`. The store fills `version`/`id` and defaults `createdAt` to now; a caller reconstructing a persisted session passes the original `createdAt` and persisted `seedLength` to preserve them. Disposed with the calling fiber.
|
||||
- `ctx.sessions.fork(source, boundary?, childSessionId?): Session` — Resolve a live session object or id, select a seed through the inclusive `boundary` event seq (default: current last event), require that boundary to be `turn/end`, and create a live child session with lineage metadata.
|
||||
- `ctx.sessions.get(id: SessionId): Session | undefined`
|
||||
- `ctx.sessions.list(): Session[]`
|
||||
|
||||
@@ -54,7 +55,7 @@ The `request/header` (full `EpochHeader` snapshot with a `RequestHeaderReason`)
|
||||
|
||||
### Session event vocabulary (`types.ts`)
|
||||
|
||||
The append-only log's event types, enumerated member by member — payloads, surface badges, provenance — in the generated [persistence log event catalog](../../../docs/persistence-catalog/log-events.md). Token usage rides on `assistant/message.usage`; an operational error's step is on `turn/end.reason` for `kind: 'error'`.
|
||||
The append-only log's event types, enumerated member by member — payloads, surface badges, provenance — in the generated [persistence log event catalog](../../../docs/persistence-catalog.md). Token usage rides on `assistant/message.usage`; an operational error's step is on `turn/end.reason` for `kind: 'error'`.
|
||||
|
||||
Merge-extensible via `SessionEventMap` — a plugin declaration-merges its own types (the compaction seam's `compact/*`, the hook bridges' `hook/*`); merged members appear in the same catalog.
|
||||
|
||||
@@ -72,9 +73,9 @@ Every `SessionEvent` carries two optional top-level fields (structural metadata)
|
||||
### Extension points
|
||||
|
||||
- Persistence plugins: subscribe to `session/event` (write-behind) and drain on `session/flush` (awaited) and fiber dispose. A durable backend reads the log and reloads it into a live session; the metadata seam (`SessionHeader`, `session.header`) is what such a backend stores beside the log.
|
||||
- Replay/fork: `ctx.sessions.create(id, { seed })` seeds a new session with an existing event log. The surface rebuilds deterministically from `surfaceOp` markers in the seeded events. The seed is validated to the SAME invariants `append` enforces — including that every surface-eligible event (`SurfaceEventType`) carries a `surfaceOp` marker — so a marker-less message event is rejected at construction rather than silently vanishing from `deriveMessages()` (the surface is the sole derivation path) on resume.
|
||||
- Replay/fork: `ctx.sessions.create(id, { seed })` seeds a new session with an existing event log. The surface rebuilds deterministically from `surfaceOp` markers in the seeded events. The seed is validated to the SAME always-on invariants `append` enforces — contiguous seqs, JSON-serializable data, and required `surfaceOp` markers on surface-eligible events — so marker-less message events are rejected at construction rather than silently vanishing from `deriveMessages()`. Broader turn-enclosure checks stay in `dsh-invariants` and persistence repair. Ordinary live-session forks use `ctx.sessions.fork(source, boundary?, childSessionId?)`, where `boundary` is the inclusive source event seq to fork through.
|
||||
- Compaction: the `dsh-compact-basic` plugin appends a `user/message` with `surfaceOp: { op: 'replace', start, end }` to shadow old surface nodes behind a summary checkpoint.
|
||||
|
||||
### What is NOT here (TODO)
|
||||
|
||||
- **Session branching/tree** (pi-style entry tree) — deferred unless needed beyond seed-based forking.
|
||||
- **Session branching/tree** (pi-style entry tree) — deferred unless needed beyond boundary-based `fork()`.
|
||||
|
||||
@@ -153,10 +153,15 @@ export class Session {
|
||||
this.header = header ?? { version: SESSION_FORMAT_VERSION, id, createdAt: Date.now() }
|
||||
}
|
||||
|
||||
/**
|
||||
* The append-only event log, exposed live by reference (readonly-typed, not
|
||||
* a snapshot): later appends are visible through the same array.
|
||||
*/
|
||||
get events(): readonly SessionEvent[] {
|
||||
return this.log
|
||||
}
|
||||
|
||||
/** The next event's sequence number — always the log length (the `seq = log.length` contiguity contract). */
|
||||
get seq(): number {
|
||||
return this.log.length
|
||||
}
|
||||
@@ -175,6 +180,9 @@ export class Session {
|
||||
* declare how it joins the surface, the sole source of derived history) and
|
||||
* rejected by the compiler for non-surface types like `turn/start` or
|
||||
* `assistant/chunk`.
|
||||
* @returns the logged event — its assigned `seq`/`time` plus the SNAPSHOT of
|
||||
* `data` that entered the log, so reading `event.data` back sees the logged
|
||||
* value, never the caller's still-mutable input.
|
||||
* @throws if `data` is not losslessly JSON-serializable (BigInt, function,
|
||||
* symbol, undefined, non-finite number, circular ref, or an exotic object
|
||||
* like Map/Set/Date). The event log is the durable source of truth, so this
|
||||
@@ -362,6 +370,32 @@ export class Session {
|
||||
}
|
||||
}
|
||||
|
||||
/** A fork source: either the live session object or its live store id. */
|
||||
export type SessionForkSource = Session | SessionId
|
||||
|
||||
/**
|
||||
* Rejection codes for session forking: the fork source id is unknown to the
|
||||
* live store (`SESSION_NOT_FOUND`) or names a session object that is not the
|
||||
* store's live instance (`SESSION_NOT_LIVE`); the requested child id is
|
||||
* already taken (`SESSION_ALREADY_EXISTS`); the boundary is not a contiguous
|
||||
* existing seq (`INVALID_BOUNDARY`); or the boundary event is not a
|
||||
* `turn/end` — a fork must cut on a closed turn (`OPEN_TURN`).
|
||||
*/
|
||||
export type SessionForkErrorCode =
|
||||
| 'SESSION_NOT_FOUND'
|
||||
| 'SESSION_NOT_LIVE'
|
||||
| 'SESSION_ALREADY_EXISTS'
|
||||
| 'INVALID_BOUNDARY'
|
||||
| 'OPEN_TURN'
|
||||
|
||||
/** Typed error for session fork rejections. */
|
||||
export class SessionForkError extends Error {
|
||||
constructor(message: string, public readonly code: SessionForkErrorCode) {
|
||||
super(message)
|
||||
this.name = 'SessionForkError'
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* In-memory session store (`ctx.sessions`).
|
||||
*
|
||||
@@ -496,6 +530,92 @@ export class SessionStore extends Service {
|
||||
list(): Session[] {
|
||||
return [...this.store.values()]
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a live child session from a turn-enclosed prefix of a live source.
|
||||
* `boundary` is an inclusive source event seq; omitted means the source's
|
||||
* current last event. A non-empty selected slice must end at `turn/end`.
|
||||
*
|
||||
* @param source - Live source session object or id.
|
||||
* @param boundary - Inclusive source event seq to fork through; omitted means
|
||||
* the source's current last event, and omitted on an empty source forks an
|
||||
* empty child.
|
||||
* @param childSessionId - Optional child session id; omitted delegates to
|
||||
* `SessionStore`'s id policy.
|
||||
* @returns The created live child session.
|
||||
*/
|
||||
fork(source: SessionForkSource, boundary?: number, childSessionId?: SessionId): Session {
|
||||
if (childSessionId !== undefined && this.get(childSessionId) !== undefined) {
|
||||
throw new SessionForkError(`session "${childSessionId}" already exists`, 'SESSION_ALREADY_EXISTS')
|
||||
}
|
||||
const liveSource = this._resolveForkSource(source)
|
||||
const seed = this._forkSeed(liveSource, boundary)
|
||||
return this.create(childSessionId, {
|
||||
seed,
|
||||
meta: {
|
||||
...liveSource.header.cwd !== undefined ? { cwd: liveSource.header.cwd } : {},
|
||||
parentSession: liveSource.id,
|
||||
seedLength: seed.length,
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
private _forkSeed(session: Session, requestedBoundary: number | undefined): SessionEvent[] {
|
||||
const events = session.events
|
||||
const lastEvent = events.at(-1)
|
||||
let boundary: number
|
||||
if (requestedBoundary !== undefined) {
|
||||
boundary = requestedBoundary
|
||||
} else {
|
||||
if (lastEvent === undefined) return []
|
||||
boundary = lastEvent.seq
|
||||
}
|
||||
if (!Number.isSafeInteger(boundary) || boundary < 0) {
|
||||
throw new SessionForkError(
|
||||
`fork boundary for session "${session.id}" must be a non-negative safe integer, got ${String(boundary)}`,
|
||||
'INVALID_BOUNDARY',
|
||||
)
|
||||
}
|
||||
if (boundary >= events.length) {
|
||||
const lastSeq = events.at(-1)?.seq
|
||||
throw new SessionForkError(
|
||||
`fork boundary ${boundary} does not exist in session "${session.id}" (last seq: ${lastSeq ?? 'none'})`,
|
||||
'INVALID_BOUNDARY',
|
||||
)
|
||||
}
|
||||
|
||||
const boundaryEvent = events[boundary]
|
||||
if (boundaryEvent === undefined || boundaryEvent.seq !== boundary) {
|
||||
throw new SessionForkError(
|
||||
`fork boundary ${boundary} does not match a contiguous event seq in session "${session.id}"`,
|
||||
'INVALID_BOUNDARY',
|
||||
)
|
||||
}
|
||||
if (boundaryEvent.type !== 'turn/end') {
|
||||
throw new SessionForkError(
|
||||
`fork boundary ${boundary} in session "${session.id}" must be turn/end, got ${boundaryEvent.type}`,
|
||||
'OPEN_TURN',
|
||||
)
|
||||
}
|
||||
|
||||
return events.slice(0, boundary + 1).map(event => structuredClone(event))
|
||||
}
|
||||
|
||||
private _resolveForkSource(source: SessionForkSource): Session {
|
||||
if (typeof source === 'string') {
|
||||
const session = this.get(source)
|
||||
if (session === undefined) throw new SessionForkError(`session "${source}" not found`, 'SESSION_NOT_FOUND')
|
||||
return session
|
||||
}
|
||||
|
||||
const live = this.get(source.id)
|
||||
if (live === undefined) {
|
||||
throw new SessionForkError(`session "${source.id}" not found`, 'SESSION_NOT_FOUND')
|
||||
}
|
||||
if (live !== source) throw new SessionForkError(`session "${source.id}" is not the live store instance`, 'SESSION_NOT_LIVE')
|
||||
return source
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
export default SessionStore
|
||||
|
||||
@@ -40,6 +40,10 @@ export type JsonValue = null | boolean | number | string | JsonValue[] | { [key:
|
||||
* hiding under a symbol/non-enumerable key cannot make the round-trip lossy.
|
||||
* Getters are invoked during the check (again as `JSON.stringify` would), so the
|
||||
* contract is for plain data records, not objects with side-effecting accessors.
|
||||
* @param value - the candidate event data to test.
|
||||
* @param seen - objects on the current descent path, for circular-reference
|
||||
* detection; the recursion threads it — callers omit it.
|
||||
* @returns true when `value` survives a JSON round-trip losslessly.
|
||||
*/
|
||||
export function isJsonValue(value: unknown, seen: Set<object> = new Set()): boolean {
|
||||
if (value === null) return true
|
||||
|
||||
@@ -54,6 +54,8 @@ import type { SessionEvent } from './types.ts'
|
||||
* Only the LAST turn can be open: the invariants plugin guarantees a `turn/end`
|
||||
* before any later `turn/start`, so an interior open turn is impossible in a
|
||||
* valid committed log. Likewise at most one step is open within that turn.
|
||||
* @param events - the loaded durable log to scan (a valid committed prefix, possibly with a crash tail).
|
||||
* @returns the synthetic closer events to append after `events`, in order; empty when the log is already balanced.
|
||||
*/
|
||||
export function interruptedTurnClosers(events: readonly SessionEvent[]): SessionEvent[] {
|
||||
let openTurn: number | null = null
|
||||
|
||||
@@ -29,6 +29,8 @@ const SURFACE_EVENT_TYPES = new Set<string>([
|
||||
* surface-eligible event that is MISSING its mandatory marker (e.g. validating
|
||||
* a seed/load log); use {@link isSurfaceEvent} to narrow to a fully-formed
|
||||
* {@link SurfaceEvent} with `surfaceOp` present.
|
||||
* @param type - the event type string to test.
|
||||
* @returns true when the type is one of the five message-producing types.
|
||||
*/
|
||||
export function isSurfaceEligibleType(type: string): boolean {
|
||||
return SURFACE_EVENT_TYPES.has(type)
|
||||
@@ -38,6 +40,8 @@ export function isSurfaceEligibleType(type: string): boolean {
|
||||
* Narrow a {@link SessionEvent} to {@link SurfaceEvent}: checks that the
|
||||
* event's `type` is surface-eligible AND that `surfaceOp` is present.
|
||||
* The narrowed type has mandatory {@link SurfaceOp}.
|
||||
* @param event - the event to narrow.
|
||||
* @returns true when the event is surface-eligible and carries its `surfaceOp` marker.
|
||||
*/
|
||||
export function isSurfaceEvent(event: SessionEvent): event is SurfaceEvent {
|
||||
if (!SURFACE_EVENT_TYPES.has(event.type)) return false
|
||||
|
||||
@@ -74,6 +74,12 @@ function nodeDelta(event: SessionEvent): number {
|
||||
* surface successor (`SurfaceNode.next`), or `null` when `end` is the tail —
|
||||
* for the cut after `end`.
|
||||
*
|
||||
* @param nodes - the surface linked list in head→tail order.
|
||||
* @param events - the session log each node's `seq` indexes into.
|
||||
* @param beforeSeq - names the cut (the node it sits immediately before);
|
||||
* `null` — or any seq not on the surface — means the after-tail cut.
|
||||
* @returns true when every `tool-call` before the cut is answered before it
|
||||
* (the unanswered-call depth at the cut is zero).
|
||||
* @throws if the surface prefix drives the unanswered-call depth negative — a
|
||||
* `tool/result` with no preceding open `tool-call` on the surface. That is a
|
||||
* corrupt surface (a structural invariant violation), surfaced loudly here
|
||||
|
||||
@@ -4,7 +4,11 @@ import type { CallId, ContentBlock, LlmCallConfig, MessageSource, StreamChunk, T
|
||||
/** Identifies one session in the store (and its persistence artifacts). */
|
||||
export type SessionId = Branded<'SessionId'>
|
||||
|
||||
/** Brand a string as a {@link SessionId}. */
|
||||
/**
|
||||
* Brand a string as a {@link SessionId}.
|
||||
* @param id - the raw session id string.
|
||||
* @returns the same string, branded (a compile-time cast — no runtime cost).
|
||||
*/
|
||||
export function SessionId(id: string): SessionId {
|
||||
return id as SessionId
|
||||
}
|
||||
@@ -102,6 +106,7 @@ export interface TurnTriggerMap {
|
||||
injection: { kind: 'injection'; source: MessageSource }
|
||||
}
|
||||
|
||||
/** The union over {@link TurnTriggerMap} — what started a turn; plugins extend it by merging variants into the map. */
|
||||
export type TurnTrigger = TurnTriggerMap[keyof TurnTriggerMap]
|
||||
|
||||
/**
|
||||
@@ -156,6 +161,7 @@ export interface TurnEndReasonMap {
|
||||
interrupted: { kind: 'interrupted' }
|
||||
}
|
||||
|
||||
/** The union over {@link TurnEndReasonMap} — why a turn ended; plugins extend it by merging variants into the map. */
|
||||
export type TurnEndReason = TurnEndReasonMap[keyof TurnEndReasonMap]
|
||||
|
||||
/**
|
||||
@@ -361,6 +367,7 @@ export interface SessionEventMap {
|
||||
'request/header-delta': { system?: SystemDelta; tools?: ToolsDelta; config?: LlmCallConfig }
|
||||
}
|
||||
|
||||
/** The appendable event-type keys of {@link SessionEventMap}, plugin-merged extensions included. */
|
||||
export type SessionEventType = keyof SessionEventMap
|
||||
|
||||
/**
|
||||
|
||||
240
packages/core/session/tests/fork.spec.ts
Normal file
240
packages/core/session/tests/fork.spec.ts
Normal file
@@ -0,0 +1,240 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import { CallId } from '@deepseek-ai/dsh-llm'
|
||||
import SessionStore, { Session, SessionForkError, SessionId } from '@deepseek-ai/dsh-session'
|
||||
import type { SessionEvent, TurnEndReason } from '@deepseek-ai/dsh-session'
|
||||
|
||||
async function setup(): Promise<{ ctx: Context; sessions: SessionStore }> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SessionStore)
|
||||
return { ctx, sessions: ctx.sessions }
|
||||
}
|
||||
|
||||
function appendClosedTurn(
|
||||
session: Session,
|
||||
turn: number,
|
||||
text = `hello ${turn}`,
|
||||
reason: TurnEndReason = { kind: 'completed' },
|
||||
): void {
|
||||
session.append('turn/start', { turn, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('user/message', {
|
||||
content: [{ type: 'text', text }],
|
||||
source: { kind: 'user' },
|
||||
}, { surfaceOp: 'append' })
|
||||
session.append('turn/end', { turn, reason })
|
||||
}
|
||||
|
||||
function appendOpenTurn(session: Session, turn: number): void {
|
||||
session.append('turn/start', { turn, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('user/message', {
|
||||
content: [{ type: 'text', text: `open ${turn}` }],
|
||||
source: { kind: 'user' },
|
||||
}, { surfaceOp: 'append' })
|
||||
}
|
||||
|
||||
function firstUserMessage(events: readonly SessionEvent[]): SessionEvent<'user/message'> {
|
||||
const event = events.find((e): e is SessionEvent<'user/message'> => e.type === 'user/message')
|
||||
if (event === undefined) throw new Error('missing user/message')
|
||||
return event
|
||||
}
|
||||
|
||||
function lastSeq(session: Session): number {
|
||||
const event = session.events.at(-1)
|
||||
if (event === undefined) throw new Error('missing last event')
|
||||
return event.seq
|
||||
}
|
||||
|
||||
describe('SessionStore.fork', () => {
|
||||
it('forks an empty live session as an empty child with lineage metadata', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const source = ctx.sessions.create(SessionId('empty-parent'), { meta: { cwd: '/workspace' } })
|
||||
|
||||
const child = sessions.fork(source, undefined, SessionId('empty-child'))
|
||||
|
||||
expect(child.events).toEqual([])
|
||||
expect(child.header).toMatchObject({
|
||||
id: SessionId('empty-child'),
|
||||
cwd: '/workspace',
|
||||
parentSession: SessionId('empty-parent'),
|
||||
seedLength: 0,
|
||||
})
|
||||
})
|
||||
|
||||
it('forks the latest completed boundary by default and deep-clones seed events', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const source = ctx.sessions.create(SessionId('parent'), { meta: { cwd: '/workspace' } })
|
||||
appendClosedTurn(source, 1, 'hello')
|
||||
|
||||
const child = sessions.fork(SessionId('parent'), undefined, SessionId('child'))
|
||||
|
||||
expect(child.events).toEqual(source.events)
|
||||
expect(child.events).not.toBe(source.events)
|
||||
expect(child.events[1]).not.toBe(source.events[1])
|
||||
firstUserMessage(child.events).data.content[0] = { type: 'text', text: 'child mutation' }
|
||||
expect(firstUserMessage(source.events).data.content).toEqual([{ type: 'text', text: 'hello' }])
|
||||
expect(child.header).toMatchObject({
|
||||
id: SessionId('child'),
|
||||
cwd: '/workspace',
|
||||
parentSession: SessionId('parent'),
|
||||
seedLength: source.events.length,
|
||||
})
|
||||
})
|
||||
|
||||
it('forks from an earlier turn boundary even when the source currently has an open tail', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const source = ctx.sessions.create(SessionId('parent'), { meta: { cwd: '/workspace' } })
|
||||
appendClosedTurn(source, 1, 'first')
|
||||
const firstBoundary = lastSeq(source)
|
||||
appendClosedTurn(source, 2, 'second')
|
||||
appendOpenTurn(source, 3)
|
||||
|
||||
const child = sessions.fork(source, firstBoundary, SessionId('child-from-first'))
|
||||
|
||||
expect(child.events).toEqual(source.events.slice(0, firstBoundary + 1))
|
||||
expect(child.header.seedLength).toBe(firstBoundary + 1)
|
||||
expect(child.deriveMessages()).toEqual([{ role: 'user', content: [{ type: 'text', text: 'first' }] }])
|
||||
})
|
||||
|
||||
it('accepts every turn/end reason as an explicit fork boundary', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const reasons: TurnEndReason[] = [
|
||||
{ kind: 'completed' },
|
||||
{ kind: 'aborted', reason: 'cancelled by user' },
|
||||
{ kind: 'error', step: 1, message: 'model failed', code: 'MODEL' },
|
||||
{ kind: 'disposed' },
|
||||
{ kind: 'max-tokens' },
|
||||
{ kind: 'interrupted' },
|
||||
]
|
||||
|
||||
for (const reason of reasons) {
|
||||
const source = ctx.sessions.create(SessionId(`parent-${reason.kind}`))
|
||||
appendClosedTurn(source, 1, reason.kind, reason)
|
||||
|
||||
const child = sessions.fork(source, lastSeq(source), SessionId(`child-${reason.kind}`))
|
||||
|
||||
expect(child.events.at(-1)?.type).toBe('turn/end')
|
||||
expect(child.header.seedLength).toBe(source.events.length)
|
||||
}
|
||||
})
|
||||
|
||||
it('rejects invalid boundaries before creating a child', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const empty = ctx.sessions.create(SessionId('empty'))
|
||||
expect(() => sessions.fork(empty, 0, SessionId('empty-child')))
|
||||
.toThrow(new SessionForkError('fork boundary 0 does not exist in session "empty" (last seq: none)', 'INVALID_BOUNDARY'))
|
||||
expect(ctx.sessions.get(SessionId('empty-child'))).toBeUndefined()
|
||||
|
||||
const source = ctx.sessions.create(SessionId('parent'))
|
||||
appendClosedTurn(source, 1)
|
||||
expect(() => sessions.fork(source, -1, SessionId('negative')))
|
||||
.toThrow(/non-negative safe integer/)
|
||||
expect(() => sessions.fork(source, 0.5, SessionId('fraction')))
|
||||
.toThrow(/non-negative safe integer/)
|
||||
expect(() => sessions.fork(source, Number.MAX_SAFE_INTEGER + 1, SessionId('unsafe')))
|
||||
.toThrow(/non-negative safe integer/)
|
||||
expect(() => sessions.fork(source, source.seq, SessionId('past-end')))
|
||||
.toThrow(new SessionForkError(`fork boundary ${source.seq} does not exist in session "parent" (last seq: ${source.seq - 1})`, 'INVALID_BOUNDARY'))
|
||||
})
|
||||
|
||||
it('rejects a corrupted live source whose array index no longer matches event seq', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const source = ctx.sessions.create(SessionId('corrupt-parent'))
|
||||
appendClosedTurn(source, 1)
|
||||
const mutableLog = (source as unknown as { log: SessionEvent[] }).log
|
||||
mutableLog[2] = { ...mutableLog[2]!, seq: 99 }
|
||||
|
||||
expect(() => sessions.fork(source, 2, SessionId('corrupt-child')))
|
||||
.toThrow(new SessionForkError('fork boundary 2 does not match a contiguous event seq in session "corrupt-parent"', 'INVALID_BOUNDARY'))
|
||||
expect(ctx.sessions.get(SessionId('corrupt-child'))).toBeUndefined()
|
||||
})
|
||||
|
||||
it('rejects an unknown live session id', async () => {
|
||||
const { sessions } = await setup()
|
||||
|
||||
expect(() => sessions.fork(SessionId('missing')))
|
||||
.toThrow(new SessionForkError('session "missing" not found', 'SESSION_NOT_FOUND'))
|
||||
})
|
||||
|
||||
it('rejects a detached Session object that is not live in ctx.sessions', async () => {
|
||||
const { sessions } = await setup()
|
||||
const detached = new Session(SessionId('detached'))
|
||||
|
||||
expect(() => sessions.fork(detached))
|
||||
.toThrow(new SessionForkError('session "detached" not found', 'SESSION_NOT_FOUND'))
|
||||
})
|
||||
|
||||
it('rejects a stale Session object whose id is live on a different instance', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
ctx.sessions.create(SessionId('same-id'))
|
||||
const stale = new Session(SessionId('same-id'))
|
||||
|
||||
expect(() => sessions.fork(stale))
|
||||
.toThrow(new SessionForkError('session "same-id" is not the live store instance', 'SESSION_NOT_LIVE'))
|
||||
})
|
||||
|
||||
it('rejects selected slices whose boundary is inside an open turn', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const cases: [string, (session: Session) => number][] = [
|
||||
['turn/start', (session) => {
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
return lastSeq(session)
|
||||
}],
|
||||
['step/start', (session) => {
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
return lastSeq(session)
|
||||
}],
|
||||
['user/message', (session) => {
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('user/message', { content: [{ type: 'text', text: 'open' }], source: { kind: 'user' } }, { surfaceOp: 'append' })
|
||||
return lastSeq(session)
|
||||
}],
|
||||
['assistant/message', (session) => {
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
session.append('assistant/message', { turn: 1, step: 1, content: [{ type: 'text', text: 'partial' }] }, { surfaceOp: 'append' })
|
||||
return lastSeq(session)
|
||||
}],
|
||||
['tool/call', (session) => {
|
||||
const callId = CallId('call-open')
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
session.append('assistant/message', {
|
||||
turn: 1,
|
||||
step: 1,
|
||||
content: [{ type: 'tool-call', id: callId, name: 'bash', arguments: '{}' }],
|
||||
}, { surfaceOp: 'append' })
|
||||
session.append('tool/call', { turn: 1, step: 1, callId, name: 'bash', arguments: '{}' })
|
||||
return lastSeq(session)
|
||||
}],
|
||||
]
|
||||
|
||||
for (const [lastType, build] of cases) {
|
||||
const source = ctx.sessions.create(SessionId(`open-${lastType}`))
|
||||
const boundary = build(source)
|
||||
|
||||
expect(() => sessions.fork(source, boundary))
|
||||
.toThrow(new SessionForkError(`fork boundary ${boundary} in session "open-${lastType}" must be turn/end, got ${lastType}`, 'OPEN_TURN'))
|
||||
}
|
||||
})
|
||||
|
||||
it('rejects a child session id that is already live with a typed fork error', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const source = ctx.sessions.create(SessionId('parent'))
|
||||
appendClosedTurn(source, 1)
|
||||
ctx.sessions.create(SessionId('child'))
|
||||
|
||||
expect(() => sessions.fork(source, undefined, SessionId('child')))
|
||||
.toThrow(new SessionForkError('session "child" already exists', 'SESSION_ALREADY_EXISTS'))
|
||||
})
|
||||
|
||||
it('rejects a duplicate child session id before validating the boundary', async () => {
|
||||
const { ctx, sessions } = await setup()
|
||||
const source = ctx.sessions.create(SessionId('open-parent'))
|
||||
source.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
ctx.sessions.create(SessionId('child'))
|
||||
|
||||
expect(() => sessions.fork(source, undefined, SessionId('child')))
|
||||
.toThrow(new SessionForkError('session "child" already exists', 'SESSION_ALREADY_EXISTS'))
|
||||
})
|
||||
})
|
||||
@@ -61,8 +61,10 @@ describe('Session', () => {
|
||||
|
||||
it('replays identically from a seeded event log', () => {
|
||||
const original = new Session(SessionId('s3'))
|
||||
original.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
original.append('user/message', { content: [{ type: 'text', text: 'q' }], source: { kind: 'user' } }, { surfaceOp: 'append' })
|
||||
original.append('assistant/message', { turn: 1, step: 1, content: [{ type: 'text', text: 'a' }] }, { surfaceOp: 'append' })
|
||||
original.append('turn/end', { turn: 1, reason: { kind: 'completed' } })
|
||||
|
||||
const replayed = new Session(SessionId('s3-replay'), [...original.events])
|
||||
expect(replayed.deriveMessages()).toEqual(original.deriveMessages())
|
||||
@@ -423,7 +425,9 @@ describe('todo/write event', () => {
|
||||
|
||||
it('round-trips through a seeded replay identically (durable, no surfaceOp needed)', () => {
|
||||
const original = new Session(SessionId('t4'))
|
||||
original.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
original.append('todo/write', { todos: [{ content: 'only', status: 'completed' }] })
|
||||
original.append('turn/end', { turn: 1, reason: { kind: 'completed' } })
|
||||
// Seeding a non-surface event with no surfaceOp must not throw.
|
||||
const replayed = new Session(SessionId('t4-replay'), [...original.events])
|
||||
expect(replayed.events.findLast(e => e.type === 'todo/write')!.data.todos)
|
||||
|
||||
@@ -7,15 +7,16 @@ System prompt assembly registry. Plugins contribute ordered text sections, tool-
|
||||
| Key | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `persona` | `''` | The deployment persona: the ONE deployment-authored prompt fragment, rendered as the order-0 `deployment:persona` section and shared by every agent in the context (subagents included). A template — complete `{{…}}` groups are interpreted strictly against the registered variables (the shipped loop registers `{{model}}`/`{{cwd}}`), with no escape syntax for literal braces yet. Empty ⇒ the section is dropped at render. |
|
||||
| `toolOrder` | — | Explicit model-facing tool order, as a list of `ToolSchema.name`s with one `'<unlisted-tools>'` rest entry (`TOOL_ORDER_REST`): listed tools take their listed position, unlisted tools land at the rest entry in lexicographic name order. Absent ⇒ plain lexicographic name order. Applied to the collected tools BEFORE the `system-prompt/assemble` waterfall — like the sections' `order` sort, it canonicalizes what the registry contributed (registration order is a plugin-load artifact), and a waterfall listener that mutates the list owns the determinism of what it emits. Misconfiguration fails loud: a list without exactly one rest entry, or with duplicates, throws at load; a listed name with no registered tool rejects every `assemble()`; a tool provider returning the reserved rest-entry name also rejects. Under the shipped loop the turn fails before any model request. Why a central list and not per-plugin weights: [Explicit model-facing tool order](../../../docs/rfc/implemented/feature/2026-07-06-explicit-tool-order.md). |
|
||||
|
||||
## Service: `SystemPrompt` (ctx key: `systemPrompt`)
|
||||
|
||||
### Public API
|
||||
|
||||
- `ctx.systemPrompt.section(section: PromptSection): () => void` Contribute a section. Duplicate names throw. Disposed with the calling fiber.
|
||||
- `ctx.systemPrompt.tools(provider: () => ToolSchema[]): () => void` Contribute tool schemas (evaluated at each assembly). Disposed with the calling fiber.
|
||||
- `ctx.systemPrompt.tools(provider: () => ToolSchema[]): () => void` Contribute tool schemas (evaluated at each assembly). A provider must not return a schema named `TOOL_ORDER_REST`; that name is reserved for `toolOrder`'s rest entry. Disposed with the calling fiber.
|
||||
- `ctx.systemPrompt.variable(name: string, provider: (context) => string | undefined): () => void` Contribute a prompt variable, referenced from section text as `{{name}}`. Duplicate or unreferenceable names throw; `undefined` means "no value for this assembly". Disposed with the calling fiber.
|
||||
- `ctx.systemPrompt.assemble(context?: AssembleContext): Promise<PromptAssembly>` Assemble the prompt for one caller. Runs through the `system-prompt/assemble` waterfall.
|
||||
- `ctx.systemPrompt.assemble(context?: AssembleContext): Promise<PromptAssembly>` Assemble the prompt for one caller. Runs through the `system-prompt/assemble` waterfall. Rejects when a configured `toolOrder` names a tool no provider contributed, or when a provider returns the reserved rest-entry name.
|
||||
|
||||
### Events
|
||||
|
||||
|
||||
@@ -88,7 +88,8 @@ export interface AssembledSection {
|
||||
*
|
||||
* Tool schemas are part of the assembly by design: "what the model is told it
|
||||
* can do" is one coherent thing managed here, even though adapters transmit
|
||||
* `tools` as a separate wire field rather than prompt text.
|
||||
* `tools` as a separate wire field rather than prompt text. They arrive in
|
||||
* the canonical model-facing order (see {@link Config.toolOrder}).
|
||||
*
|
||||
* `variables` carries every registered prompt variable resolved against this
|
||||
* assembly's context — key present means registered, `undefined` value means
|
||||
@@ -110,6 +111,71 @@ const VARIABLE_NAME = /^[a-z][a-z0-9_]*$/
|
||||
/** A complete `{{...}}` reference group at the scan position (validated after). */
|
||||
const GROUP_AT = /^\{\{([^{}]*)\}\}/
|
||||
|
||||
/**
|
||||
* The rest entry for {@link Config.toolOrder}: the position where registered
|
||||
* tools not named in the list are inserted (in lexicographic name order).
|
||||
* Reserved: collected tool schemas using this name are rejected before
|
||||
* ordering, so the marker can never collide with a real model-facing tool.
|
||||
*/
|
||||
export const TOOL_ORDER_REST = '<unlisted-tools>'
|
||||
|
||||
/**
|
||||
* Validate a configured tool-order list's shape at service construction:
|
||||
* the {@link TOOL_ORDER_REST} rest entry exactly once, no duplicate names.
|
||||
* Returns the list (or undefined when unconfigured); throws otherwise,
|
||||
* failing the service at load — a bad order config must never reach an
|
||||
* assembly. Whether every listed name matches a registered tool is checked
|
||||
* at each assembly instead ({@link orderTools}): tool plugins register after
|
||||
* this service constructs, so the tool set does not exist yet here.
|
||||
*/
|
||||
function validateToolOrder(toolOrder: string[] | undefined): string[] | undefined {
|
||||
if (toolOrder === undefined) return undefined
|
||||
const seen = new Set<string>()
|
||||
for (const name of toolOrder) {
|
||||
if (seen.has(name)) throw new Error(`toolOrder lists "${name}" more than once`)
|
||||
seen.add(name)
|
||||
}
|
||||
if (!seen.has(TOOL_ORDER_REST)) {
|
||||
throw new Error(`toolOrder must contain the "${TOOL_ORDER_REST}" rest entry (where unlisted tools are inserted)`)
|
||||
}
|
||||
return toolOrder
|
||||
}
|
||||
|
||||
/**
|
||||
* Order collected tool schemas by the validated policy: with no configured
|
||||
* list, plain lexicographic name order; with one, listed names take their
|
||||
* listed position and every unlisted tool lands at the
|
||||
* {@link TOOL_ORDER_REST} rest entry in lexicographic name order. A listed
|
||||
* name with no collected tool throws — misconfiguration fails loud, and this
|
||||
* is the earliest moment the registered tool set exists to check against
|
||||
* (tool plugins register after the service constructs, so load time is too
|
||||
* early): the assembly rejects, failing the caller's turn before any model
|
||||
* request. Never drops a tool, and both sorts are stable, so tools sharing a
|
||||
* name keep their collection order.
|
||||
*/
|
||||
function orderTools(tools: ToolSchema[], toolOrder: string[] | undefined): ToolSchema[] {
|
||||
const reserved = tools.find(tool => tool.name === TOOL_ORDER_REST)
|
||||
if (reserved !== undefined) {
|
||||
throw new Error(`tool provider returned reserved tool name "${TOOL_ORDER_REST}" (reserved for toolOrder's rest entry)`)
|
||||
}
|
||||
if (toolOrder === undefined) return tools.sort(compareToolNames)
|
||||
const registered = new Set(tools.map(tool => tool.name))
|
||||
const unknown = toolOrder.filter(name => name !== TOOL_ORDER_REST && !registered.has(name))
|
||||
if (unknown.length > 0) {
|
||||
throw new Error(`toolOrder lists unregistered tool${unknown.length > 1 ? 's' : ''} ${unknown.map(name => `"${name}"`).join(', ')}; registered tools: ${[...registered].sort().join(', ') || '(none)'}`)
|
||||
}
|
||||
const listed = new Set(toolOrder)
|
||||
const rest = tools.filter(tool => !listed.has(tool.name)).sort(compareToolNames)
|
||||
return toolOrder.flatMap(name =>
|
||||
name === TOOL_ORDER_REST ? rest : tools.filter(tool => tool.name === name))
|
||||
}
|
||||
|
||||
/** Lexicographic (code-unit) name comparison — locale-independent, so the order is identical on every machine. */
|
||||
function compareToolNames(a: ToolSchema, b: ToolSchema): number {
|
||||
return a.name < b.name ? -1 : a.name > b.name ? 1 : 0
|
||||
}
|
||||
|
||||
/** Plugin config: the deployment-authored fragment of the system prompt (see {@link Config.persona} for its contract). */
|
||||
export interface Config {
|
||||
/**
|
||||
* The deployment's persona — the ONE deployment-authored fragment of the
|
||||
@@ -124,6 +190,29 @@ export interface Config {
|
||||
* deployment opens with the harness identity alone.
|
||||
*/
|
||||
persona?: string
|
||||
/**
|
||||
* Explicit model-facing tool order, as a list of `ToolSchema.name`s: listed
|
||||
* tools take their listed position, and tools absent from the list are
|
||||
* inserted at the {@link TOOL_ORDER_REST} (`'<unlisted-tools>'`) entry in
|
||||
* lexicographic name order. A configured list must contain the rest entry
|
||||
* exactly once, no duplicate names, and no name without a registered tool —
|
||||
* a misconfigured order blocks work instead of silently reaching a model
|
||||
* request: shape violations throw at load, and an unregistered name rejects
|
||||
* every assembly. `TOOL_ORDER_REST` is reserved for the list marker and may
|
||||
* not be a collected tool name; such a provider output also rejects the
|
||||
* assembly. The single assembly-time validation rejects either failure
|
||||
* before any model request — the earliest moment the registered tool set
|
||||
* exists to check against, since tool plugins register after this service
|
||||
* constructs. When omitted, tools are ordered lexicographically by name.
|
||||
* Applied to the tools
|
||||
* {@link SystemPrompt.assemble} collects, BEFORE the
|
||||
* `system-prompt/assemble` waterfall — like the sections' `order` sort, it
|
||||
* canonicalizes what the registry contributed (registration order is a
|
||||
* plugin-load artifact); a waterfall listener that mutates the tool list
|
||||
* owns the determinism of what it emits. Rationale (and why not per-plugin
|
||||
* weights): docs/rfc/implemented/feature/2026-07-06-explicit-tool-order.md.
|
||||
*/
|
||||
toolOrder?: string[]
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -138,6 +227,10 @@ export interface Config {
|
||||
* while a `}}` still follows (e.g. `{{{model}}}`, `{{a{b}}`) all throw. A
|
||||
* lone `{{` with no `}}` anywhere after it is ordinary prose and passes
|
||||
* through verbatim. Substituted values are never re-scanned.
|
||||
* @param assembly - the assembly to render (typically the awaited result of
|
||||
* {@link SystemPrompt.assemble}); only `sections` and `variables` are read.
|
||||
* @returns the full system prompt text; `''` when every section renders empty
|
||||
* (the caller then sends no system prompt at all).
|
||||
*/
|
||||
export function renderPrompt(assembly: PromptAssembly): string {
|
||||
return assembly.sections
|
||||
@@ -198,14 +291,23 @@ function interpolate(section: AssembledSection, variables: Record<string, string
|
||||
export class SystemPrompt extends Service {
|
||||
static Config: z<Config> = z.object({
|
||||
persona: z.string().default(''),
|
||||
// A schemastery array defaults to [] when omitted, but an omitted
|
||||
// toolOrder must stay absent ("lexicographic order"), not become an
|
||||
// explicitly-configured empty list (which is invalid — it lacks the
|
||||
// rest entry). Forcing the default to undefined keeps the key out of the
|
||||
// validated config; the cast is needed because .default() expects the
|
||||
// array type.
|
||||
toolOrder: z.array(z.string()).default(undefined as unknown as string[]),
|
||||
})
|
||||
|
||||
private sections: PromptSection[] = []
|
||||
private toolProviders: (() => ToolSchema[])[] = []
|
||||
private variableProviders = new Map<string, (context: AssembleContext) => string | undefined>()
|
||||
private readonly toolOrder: string[] | undefined
|
||||
|
||||
constructor(ctx: Context, public config: Config) {
|
||||
super(ctx, 'systemPrompt')
|
||||
this.toolOrder = validateToolOrder(config.toolOrder)
|
||||
// The harness-owned openers. They live HERE (not on the loop plugin) so a
|
||||
// deployment that swaps in a different loop keeps them: the identity is a
|
||||
// harness fact stated ahead of everything, and the persona is the
|
||||
@@ -261,7 +363,10 @@ export class SystemPrompt extends Service {
|
||||
/**
|
||||
* Contribute a tool-schema provider that is evaluated at each assembly
|
||||
* call (so it can reflect the live registry state). The provider is
|
||||
* removed when the calling fiber is disposed. Emits `system-prompt/change`.
|
||||
* removed when the calling fiber is disposed. A provider must not return a
|
||||
* schema named {@link TOOL_ORDER_REST}; that name is reserved for
|
||||
* {@link Config.toolOrder}'s rest entry and rejects the assembly. Emits
|
||||
* `system-prompt/change`.
|
||||
* @param provider - evaluated at every {@link assemble} for fresh schemas.
|
||||
* @returns the disposer that removes the provider.
|
||||
*/
|
||||
@@ -318,19 +423,28 @@ export class SystemPrompt extends Service {
|
||||
|
||||
/**
|
||||
* Assemble the current prompt for one caller: section texts are resolved
|
||||
* against `context` and sorted by order, tools collected from all
|
||||
* providers, and every registered variable resolved against `context` into
|
||||
* `assembly.variables`. Tool schemas are deep-cloned because adapters and
|
||||
* request waterfalls may mutate schema objects. Runs through the
|
||||
* `system-prompt/assemble` waterfall, giving listeners the opportunity to
|
||||
* mutate or replace the assembly before it reaches the model. Await the
|
||||
* result before reading the assembly values — waterfall listeners may be
|
||||
* async. Interpolation happens later, in {@link renderPrompt}.
|
||||
* against `context` and sorted by order, tools collected from all providers
|
||||
* and put in the canonical model-facing order ({@link Config.toolOrder}, or
|
||||
* lexicographic name order when unconfigured — provider registration order
|
||||
* is a plugin-load artifact and never reaches the assembly; a configured
|
||||
* order naming a tool no provider contributed rejects the assembly), and every
|
||||
* registered variable resolved against `context` into `assembly.variables`.
|
||||
* Tool schemas are deep-cloned because adapters and request waterfalls may
|
||||
* mutate schema objects. Runs through the `system-prompt/assemble`
|
||||
* waterfall, giving listeners the opportunity to mutate or replace the
|
||||
* assembly before it reaches the model — like the sections' `order` sort,
|
||||
* tool canonicalization happens on the initial assembly, and a listener
|
||||
* owns the determinism of whatever it emits. Await the result before
|
||||
* reading the assembly values — waterfall listeners may be async.
|
||||
* Interpolation happens later, in {@link renderPrompt}.
|
||||
* @param context - what this assembly is for (defaults to an empty context;
|
||||
* see {@link AssembleContext}).
|
||||
* @returns the assembly after the waterfall has run.
|
||||
*/
|
||||
assemble(context: AssembleContext = {}): Promise<PromptAssembly> {
|
||||
// async so the misconfigured-toolOrder throw in orderTools surfaces as a
|
||||
// rejection: a Promise-returning method must not throw synchronously
|
||||
// (`assemble().catch(...)` would miss it).
|
||||
async assemble(context: AssembleContext = {}): Promise<PromptAssembly> {
|
||||
const variables: Record<string, string | undefined> = {}
|
||||
for (const [name, provider] of this.variableProviders) {
|
||||
variables[name] = provider(context)
|
||||
@@ -343,8 +457,10 @@ export class SystemPrompt extends Service {
|
||||
text: typeof section.text === 'function' ? section.text(context) : section.text,
|
||||
}))
|
||||
.sort((a, b) => a.order - b.order),
|
||||
tools: this.toolProviders.flatMap(provider =>
|
||||
provider().map(tool => ({ ...tool, parameters: structuredClone(tool.parameters) }))),
|
||||
tools: orderTools(
|
||||
this.toolProviders.flatMap(provider =>
|
||||
provider().map(tool => ({ ...tool, parameters: structuredClone(tool.parameters) }))),
|
||||
this.toolOrder),
|
||||
variables,
|
||||
}
|
||||
return this.ctx.waterfall(this, 'system-prompt/assemble', assembly, context, () => Promise.resolve(assembly))
|
||||
|
||||
115
packages/core/system-prompt/tests/tool-order.spec.ts
Normal file
115
packages/core/system-prompt/tests/tool-order.spec.ts
Normal file
@@ -0,0 +1,115 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import SystemPrompt, { PromptAssembly, TOOL_ORDER_REST } from '@deepseek-ai/dsh-system-prompt'
|
||||
import type { ToolSchema } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
function tool(name: string, description = name): ToolSchema {
|
||||
return { name, description, parameters: { type: 'object', properties: {} } }
|
||||
}
|
||||
|
||||
async function mount(config: { persona?: string; toolOrder?: string[] } = {}): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt, config)
|
||||
return ctx
|
||||
}
|
||||
|
||||
function names(assembly: PromptAssembly): string[] {
|
||||
return assembly.tools.map(t => t.name)
|
||||
}
|
||||
|
||||
describe('SystemPrompt tool order', () => {
|
||||
// The ONE place the public constant's value is pinned; everything else
|
||||
// (tests and deployment configs alike) references TOOL_ORDER_REST.
|
||||
it('exports the rest entry as "<unlisted-tools>"', () => {
|
||||
expect(TOOL_ORDER_REST).toBe('<unlisted-tools>')
|
||||
})
|
||||
|
||||
it('assembles tools in lexicographic name order when no toolOrder is configured', async () => {
|
||||
const ctx = await mount()
|
||||
ctx.systemPrompt.tools(() => [tool('charlie'), tool('alpha')])
|
||||
ctx.systemPrompt.tools(() => [tool('bravo')])
|
||||
expect(names(await ctx.systemPrompt.assemble())).toEqual(['alpha', 'bravo', 'charlie'])
|
||||
})
|
||||
|
||||
it('assembles the same order regardless of provider registration order', async () => {
|
||||
const forward = await mount()
|
||||
forward.systemPrompt.tools(() => [tool('alpha')])
|
||||
forward.systemPrompt.tools(() => [tool('zulu')])
|
||||
const backward = await mount()
|
||||
backward.systemPrompt.tools(() => [tool('zulu')])
|
||||
backward.systemPrompt.tools(() => [tool('alpha')])
|
||||
expect(names(await forward.systemPrompt.assemble())).toEqual(['alpha', 'zulu'])
|
||||
expect(names(await backward.systemPrompt.assemble())).toEqual(['alpha', 'zulu'])
|
||||
})
|
||||
|
||||
it('applies a configured toolOrder: listed positions, rest at the rest entry lexicographically', async () => {
|
||||
const ctx = await mount({ toolOrder: ['todo_write', TOOL_ORDER_REST, 'bash'] })
|
||||
ctx.systemPrompt.tools(() => [tool('bash'), tool('echo_b'), tool('todo_write'), tool('echo_a')])
|
||||
expect(names(await ctx.systemPrompt.assemble())).toEqual(['todo_write', 'echo_a', 'echo_b', 'bash'])
|
||||
})
|
||||
|
||||
it('rejects the assembly when toolOrder names a tool that is not registered (misconfiguration blocks work)', async () => {
|
||||
const ctx = await mount({ toolOrder: ['todo_write', 'ghost', TOOL_ORDER_REST, 'wraith'] })
|
||||
ctx.systemPrompt.tools(() => [tool('bash'), tool('todo_write')])
|
||||
await expect(ctx.systemPrompt.assemble()).rejects.toThrow(
|
||||
'toolOrder lists unregistered tools "ghost", "wraith"; registered tools: bash, todo_write')
|
||||
})
|
||||
|
||||
it('names the single unregistered tool when no tools are registered at all', async () => {
|
||||
const ctx = await mount({ toolOrder: ['ghost', TOOL_ORDER_REST] })
|
||||
await expect(ctx.systemPrompt.assemble()).rejects.toThrow(
|
||||
'toolOrder lists unregistered tool "ghost"; registered tools: (none)')
|
||||
})
|
||||
|
||||
it.each([
|
||||
['without an explicit toolOrder', undefined],
|
||||
['with only the rest entry configured', [TOOL_ORDER_REST]],
|
||||
])('rejects a provider tool named like the reserved rest entry %s', async (_case, toolOrder) => {
|
||||
const ctx = await mount(toolOrder === undefined ? {} : { toolOrder })
|
||||
ctx.systemPrompt.tools(() => [tool(TOOL_ORDER_REST)])
|
||||
await expect(ctx.systemPrompt.assemble()).rejects.toThrow(
|
||||
`tool provider returned reserved tool name "${TOOL_ORDER_REST}"`)
|
||||
})
|
||||
|
||||
it('keeps collection order between tools that share a name (stable sort)', async () => {
|
||||
const ctx = await mount()
|
||||
ctx.systemPrompt.tools(() => [tool('dup', 'first'), tool('anchor'), tool('dup', 'second')])
|
||||
const assembly = await ctx.systemPrompt.assemble()
|
||||
expect(assembly.tools.map(t => t.description)).toEqual(['anchor', 'first', 'second'])
|
||||
})
|
||||
|
||||
it('canonicalizes BEFORE the assemble waterfall: listeners see the ordered list and own their own edits', async () => {
|
||||
const ctx = await mount()
|
||||
ctx.systemPrompt.tools(() => [tool('zulu'), tool('alpha')])
|
||||
let seen: string[] | undefined
|
||||
ctx.on('system-prompt/assemble', function (assembly, _context, next) {
|
||||
seen = assembly.tools.map(t => t.name)
|
||||
// A listener-appended tool is NOT re-sorted — same contract as sections:
|
||||
// canonicalization applies to what the registry contributed, and a
|
||||
// listener owns the determinism of what it emits.
|
||||
assembly.tools.push(tool('aardvark'))
|
||||
return next()
|
||||
})
|
||||
const assembly = await ctx.systemPrompt.assemble()
|
||||
expect(seen).toEqual(['alpha', 'zulu'])
|
||||
expect(names(assembly)).toEqual(['alpha', 'zulu', 'aardvark'])
|
||||
})
|
||||
|
||||
it.each([
|
||||
['an empty list', []],
|
||||
['a list without the rest entry', ['bash', 'todo_write']],
|
||||
])('rejects %s at load (the rest entry is required)', async (_case, toolOrder) => {
|
||||
await expect(new Context().plugin(SystemPrompt, { toolOrder })).rejects.toThrow(`must contain the "${TOOL_ORDER_REST}" rest entry`)
|
||||
})
|
||||
|
||||
it.each([
|
||||
['a duplicate tool name', ['bash', 'bash', TOOL_ORDER_REST]],
|
||||
['a duplicate rest entry', [TOOL_ORDER_REST, 'bash', TOOL_ORDER_REST]],
|
||||
])('rejects %s at load', async (_case, toolOrder) => {
|
||||
await expect(new Context().plugin(SystemPrompt, { toolOrder })).rejects.toThrow('more than once')
|
||||
})
|
||||
|
||||
it('throws from direct construction too', () => {
|
||||
expect(() => new SystemPrompt(new Context(), { toolOrder: ['bash'] })).toThrow('rest entry')
|
||||
})
|
||||
})
|
||||
@@ -8,7 +8,7 @@ Tool registry and execution pipeline. Tool plugins register their schemas and ex
|
||||
|
||||
- `ctx.tools.register(definition: ToolDefinition): () => void` Register a tool. Disposed with the calling fiber.
|
||||
- `ctx.tools.get(name: string): ToolDefinition | undefined`
|
||||
- `ctx.tools.schemas(): ToolSchema[]` Schemas of all registered tools (without the `execute` functions). The shipped tools' schemas are catalogued in [docs/tool-catalog/tools.md](../../../docs/tool-catalog/tools.md), generated by booting each tool plugin and harvesting this method (see [the tool-schema-catalog RFC](../../../docs/rfc/implemented/process/2026-07-02-tool-schema-catalog.md)).
|
||||
- `ctx.tools.schemas(): ToolSchema[]` Schemas of all registered tools (without the `execute` functions). The shipped tools' schemas are catalogued in [docs/tool-catalog.md](../../../docs/tool-catalog.md), generated by booting each tool plugin and harvesting this method (see [the tool-schema-catalog RFC](../../../docs/rfc/implemented/process/2026-07-02-tool-schema-catalog.md)).
|
||||
- `ctx.tools.execute(exec: ToolExecution): Promise<ToolExecutionResult>` Execute one tool call through the `tools/pre-execute` → `tools/execute` → `tools/post-execute` pipeline.
|
||||
|
||||
### Injected services
|
||||
@@ -72,6 +72,12 @@ A `defineTool` tool also **validates the model-generated arguments against its `
|
||||
|
||||
See `defineTool`, `validateArgs`, `ToolArgsError`, `SchemaSpec`, `InferArgs`, and `schemaSpecToJsonSchema` in the public API for details.
|
||||
|
||||
### Structured-output schema subset
|
||||
|
||||
A separate vocabulary for callers that DEMAND a machine-readable value from an agent — the subagent seam's `SubagentStartRequest.outputSchema` (and, by extension, a workflow's `agent({ schema })`). Unlike `SchemaSpec` (the author-facing DSL for tool parameters), a `StructuredOutputSchema` is an object-rooted **raw JSON Schema subset** as data: it travels verbatim to the model as a forced tool's `parameters`, and the produced value is validated against it.
|
||||
|
||||
The subset is deliberately narrow and REJECTS LOUD outside it — accepting a keyword the validator doesn't enforce would validate less than the schema promises (accepted-then-ignored). Supported: single-string `type` (`object`/`array`/`string`/`number`/`integer`/`boolean`/`null`; type arrays rejected), `properties`/`required`/`additionalProperties` (boolean; every `required` key must be declared), `items`, scalar-only `enum`/`const`; annotations (`description`/`title`/`default`/`examples`) are ignored but must still be JSON data. `assertSupportedOutputSchema(schema)` throws `OutputSchemaError` (`code: 'UNSUPPORTED_SCHEMA'`, listing every violation) for anything else; `validateStructuredValue(schema, value)` returns path-qualified violations (empty = valid, total — never throws).
|
||||
|
||||
### Tool-owned UI presentation
|
||||
|
||||
A tool owns how ITS calls render in a UI (an editor's tool-call card, a CLI log line) — a UI plugin must NOT special-case tool names. A `ToolDefinition` may declare two optional, pure, display-only methods that return a **`card`-tagged render intent** (a discriminated union — a tool declares its card kind once and a UI bridge switches on `card`):
|
||||
|
||||
@@ -29,6 +29,16 @@ export {
|
||||
type JsonSchemaObject,
|
||||
} from './schema.ts'
|
||||
|
||||
export {
|
||||
assertSupportedOutputSchema,
|
||||
validateStructuredValue,
|
||||
OutputSchemaError,
|
||||
type StructuredOutputSchema,
|
||||
type StructuredSchemaNode,
|
||||
type StructuredSchemaType,
|
||||
type StructuredScalar,
|
||||
} from './json-schema.ts'
|
||||
|
||||
// The render-intent vocabulary a tool declares via `presentCall`/`presentResult`
|
||||
// lives in its own UI-facing module; re-export it so `@deepseek-ai/dsh-tools`
|
||||
// stays the single public surface for consumers (producers + the ACP bridge).
|
||||
|
||||
345
packages/core/tools/src/json-schema.ts
Normal file
345
packages/core/tools/src/json-schema.ts
Normal file
@@ -0,0 +1,345 @@
|
||||
/**
|
||||
* Structured-output JSON Schema subset: the vocabulary a caller uses to demand
|
||||
* a machine-readable result from a subagent (`SubagentStartRequest.outputSchema`)
|
||||
* or a workflow `agent()` call.
|
||||
*
|
||||
* This is deliberately NOT full JSON Schema. The schema travels verbatim to the
|
||||
* model as a forced tool's `parameters`, and the value the model produces is
|
||||
* validated here — so every accepted keyword must be one this module actually
|
||||
* enforces. Accepting a keyword we don't enforce would validate less than the
|
||||
* schema promises (accepted-then-ignored), so anything outside the subset is
|
||||
* REJECTED LOUD by {@link assertSupportedOutputSchema} instead. The subset:
|
||||
*
|
||||
* - `type` — a single string (`object`/`array`/`string`/`number`/`integer`/
|
||||
* `boolean`/`null`); type ARRAYS (`["string","null"]`) are rejected.
|
||||
* - `properties`/`required`/`additionalProperties` (boolean) on objects; every
|
||||
* `required` key must be declared in `properties`. `additionalProperties`
|
||||
* absent keeps standard JSON Schema semantics (extra keys allowed).
|
||||
* - `items` on arrays (absent ⇒ any JSON items).
|
||||
* - `enum` (non-empty, scalars only) and `const` (scalar) on scalar types.
|
||||
* - Annotations `description`/`title`/`default`/`examples` are allowed and
|
||||
* ignored (they constrain nothing), except that they must still be JSON data
|
||||
* — the schema is serialized onto the wire, so a non-JSON annotation would be
|
||||
* silently mangled.
|
||||
*
|
||||
* Values checked by {@link validateStructuredValue} are expected to be plain
|
||||
* host-realm JSON data (model tool-call arguments are parsed wire JSON; a
|
||||
* caller holding foreign-realm data materializes it first).
|
||||
*
|
||||
* @module dsh-tools/json-schema
|
||||
*/
|
||||
|
||||
import { assertNever, HarnessError } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
/** The scalar values `enum`/`const` may carry (finite numbers only). */
|
||||
export type StructuredScalar = string | number | boolean | null
|
||||
|
||||
/** The `type` keywords the subset accepts. */
|
||||
export type StructuredSchemaType = 'object' | 'array' | 'string' | 'number' | 'integer' | 'boolean' | 'null'
|
||||
|
||||
/**
|
||||
* One node of the structured-output schema subset. Recursive via `properties`
|
||||
* and `items`; see the module doc for the exact keyword semantics.
|
||||
*/
|
||||
export interface StructuredSchemaNode {
|
||||
type: StructuredSchemaType
|
||||
/** Nested property schemas (`type: 'object'` only). */
|
||||
properties?: Record<string, StructuredSchemaNode>
|
||||
/** Required property names; each must appear in `properties`. */
|
||||
required?: string[]
|
||||
/** `false` rejects undeclared keys; absent/`true` allows them (JSON Schema default). */
|
||||
additionalProperties?: boolean
|
||||
/** Item schema (`type: 'array'` only); absent ⇒ any JSON items. */
|
||||
items?: StructuredSchemaNode
|
||||
/** Allowed values (scalar types only). */
|
||||
enum?: StructuredScalar[]
|
||||
/** The single allowed value (scalar types only). */
|
||||
const?: StructuredScalar
|
||||
/** Annotation, ignored for validation. */
|
||||
description?: string
|
||||
/** Annotation, ignored for validation. */
|
||||
title?: string
|
||||
/** Annotation, ignored for validation (must still be JSON data). */
|
||||
default?: unknown
|
||||
/** Annotation, ignored for validation (must still be JSON data). */
|
||||
examples?: unknown
|
||||
}
|
||||
|
||||
/** A structured-output schema: an OBJECT-rooted {@link StructuredSchemaNode}. */
|
||||
export type StructuredOutputSchema = StructuredSchemaNode & { type: 'object' }
|
||||
|
||||
/**
|
||||
* Thrown by {@link assertSupportedOutputSchema} when a schema falls outside the
|
||||
* supported subset. Extends {@link HarnessError} (`code: 'UNSUPPORTED_SCHEMA'`)
|
||||
* so seam code and tool results can route on it; `violations` lists every
|
||||
* offending path, not just the first.
|
||||
*/
|
||||
export class OutputSchemaError extends HarnessError {
|
||||
/** The individual violation messages, in walk order. */
|
||||
readonly violations: string[]
|
||||
|
||||
constructor(violations: string[]) {
|
||||
super(`unsupported output schema: ${violations.join('; ')}`, 'UNSUPPORTED_SCHEMA')
|
||||
this.name = 'OutputSchemaError'
|
||||
this.violations = violations
|
||||
}
|
||||
}
|
||||
|
||||
/** The keywords the subset accepts, checked (`constraint`) or ignored (`annotation`). */
|
||||
const CONSTRAINT_KEYWORDS = new Set(['type', 'properties', 'required', 'additionalProperties', 'items', 'enum', 'const'])
|
||||
const ANNOTATION_KEYWORDS = new Set(['description', 'title', 'default', 'examples'])
|
||||
|
||||
const SCHEMA_TYPES: readonly StructuredSchemaType[] = ['object', 'array', 'string', 'number', 'integer', 'boolean', 'null']
|
||||
|
||||
/**
|
||||
* Whether a value is a PLAIN JSON object — non-null, non-array, and with a
|
||||
* prototype chain of at most one link (`null`-proto, or any realm's
|
||||
* `Object.prototype`, whose own prototype is `null`). Realm-agnostic on
|
||||
* purpose: a schema materialized in another realm carries THAT realm's
|
||||
* `Object.prototype`, which an identity check would wrongly reject. Exotic
|
||||
* hosts (`Date`, `Map`, class instances) have longer chains and are rejected —
|
||||
* they would serialize lossily (`Date` → string, `Map` → `{}`) instead of
|
||||
* failing loud.
|
||||
*/
|
||||
function isObjectLike(value: unknown): value is Record<string, unknown> {
|
||||
if (typeof value !== 'object' || value === null || Array.isArray(value)) return false
|
||||
const proto: unknown = Object.getPrototypeOf(value)
|
||||
return proto === null || Object.getPrototypeOf(proto) === null
|
||||
}
|
||||
|
||||
/** Whether a value is a supported scalar (`enum`/`const` member): string, finite number, boolean, or null. */
|
||||
function isStructuredScalar(value: unknown): value is StructuredScalar {
|
||||
return value === null || typeof value === 'string' || typeof value === 'boolean'
|
||||
|| (typeof value === 'number' && Number.isFinite(value))
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a value is JSON data (annotation payloads only): scalars, arrays, and
|
||||
* object-likes of such values. Realm-agnostic on purpose (no prototype check) —
|
||||
* the schema may have been materialized from another realm; structural JSON-ness
|
||||
* is what the wire needs. Cycles are rejected via `seen`.
|
||||
*/
|
||||
function isJsonData(value: unknown, seen: Set<object>): boolean {
|
||||
if (isStructuredScalar(value)) return true
|
||||
// The scalar check above already returned for null, so `object` here is a real object.
|
||||
if (typeof value !== 'object') return false
|
||||
if (seen.has(value)) return false
|
||||
seen.add(value)
|
||||
try {
|
||||
if (Array.isArray(value)) return value.every(entry => isJsonData(entry, seen))
|
||||
// A non-plain object (Date, Map, class instance) is NOT JSON data even when
|
||||
// it has no enumerable values — it would serialize lossily, not loudly.
|
||||
if (!isObjectLike(value)) return false
|
||||
return Object.values(value).every(entry => isJsonData(entry, seen))
|
||||
} finally {
|
||||
seen.delete(value)
|
||||
}
|
||||
}
|
||||
|
||||
/** Collect subset violations for one schema node (recursive walk). */
|
||||
function checkSchemaNode(node: unknown, path: string, violations: string[], seen: Set<object>): void {
|
||||
if (!isObjectLike(node)) {
|
||||
violations.push(`${path} must be a schema object`)
|
||||
return
|
||||
}
|
||||
if (seen.has(node)) {
|
||||
violations.push(`${path} is circular`)
|
||||
return
|
||||
}
|
||||
seen.add(node)
|
||||
|
||||
for (const key of Object.keys(node)) {
|
||||
if (CONSTRAINT_KEYWORDS.has(key)) continue
|
||||
if (ANNOTATION_KEYWORDS.has(key)) {
|
||||
if (!isJsonData(node[key], new Set())) violations.push(`${path}.${key} annotation must be JSON data`)
|
||||
continue
|
||||
}
|
||||
violations.push(`${path}.${key} is not a supported keyword (subset: type/properties/required/additionalProperties/items/enum/const + annotations)`)
|
||||
}
|
||||
if (typeof node.description !== 'undefined' && typeof node.description !== 'string') {
|
||||
violations.push(`${path}.description must be a string`)
|
||||
}
|
||||
if (typeof node.title !== 'undefined' && typeof node.title !== 'string') {
|
||||
violations.push(`${path}.title must be a string`)
|
||||
}
|
||||
|
||||
const type = node.type
|
||||
if (typeof type !== 'string' || !(SCHEMA_TYPES as readonly unknown[]).includes(type)) {
|
||||
violations.push(Array.isArray(type)
|
||||
? `${path}.type must be a single type string (type arrays are not supported)`
|
||||
: `${path}.type must be one of ${SCHEMA_TYPES.join('/')}`)
|
||||
seen.delete(node)
|
||||
return
|
||||
}
|
||||
const schemaType = type as StructuredSchemaType
|
||||
|
||||
// Keywords that only make sense on one type are rejected elsewhere — an
|
||||
// `items` on an object (or `properties` on a string) is a schema-author bug
|
||||
// the subset surfaces rather than ignores.
|
||||
const allowedFor: Record<string, StructuredSchemaType[]> = {
|
||||
properties: ['object'],
|
||||
required: ['object'],
|
||||
additionalProperties: ['object'],
|
||||
items: ['array'],
|
||||
enum: ['string', 'number', 'integer', 'boolean', 'null'],
|
||||
const: ['string', 'number', 'integer', 'boolean', 'null'],
|
||||
}
|
||||
for (const [key, types] of Object.entries(allowedFor)) {
|
||||
if (key in node && !types.includes(schemaType)) {
|
||||
violations.push(`${path}.${key} is not supported on type "${schemaType}"`)
|
||||
}
|
||||
}
|
||||
|
||||
switch (schemaType) {
|
||||
case 'object': {
|
||||
const properties = node.properties
|
||||
if (properties !== undefined) {
|
||||
if (!isObjectLike(properties)) {
|
||||
violations.push(`${path}.properties must be an object of schemas`)
|
||||
} else {
|
||||
for (const [key, child] of Object.entries(properties)) {
|
||||
checkSchemaNode(child, `${path}.properties.${key}`, violations, seen)
|
||||
}
|
||||
}
|
||||
}
|
||||
const required = node.required
|
||||
if (required !== undefined) {
|
||||
if (!Array.isArray(required) || required.some(entry => typeof entry !== 'string')) {
|
||||
violations.push(`${path}.required must be an array of strings`)
|
||||
} else {
|
||||
const declared = isObjectLike(properties) ? properties : {}
|
||||
// The guard above proved every entry is a string.
|
||||
for (const key of required as string[]) {
|
||||
// Own-property check: `in` would let inherited names (`toString`)
|
||||
// satisfy the declared-in-properties contract via the prototype.
|
||||
if (!Object.hasOwn(declared, key)) violations.push(`${path}.required names "${key}" which is not in properties`)
|
||||
}
|
||||
}
|
||||
}
|
||||
if (node.additionalProperties !== undefined && typeof node.additionalProperties !== 'boolean') {
|
||||
violations.push(`${path}.additionalProperties must be a boolean`)
|
||||
}
|
||||
break
|
||||
}
|
||||
case 'array': {
|
||||
if (node.items !== undefined) checkSchemaNode(node.items, `${path}.items`, violations, seen)
|
||||
break
|
||||
}
|
||||
case 'string':
|
||||
case 'number':
|
||||
case 'integer':
|
||||
case 'boolean':
|
||||
case 'null': {
|
||||
const allowed = node.enum
|
||||
if (allowed !== undefined) {
|
||||
if (!Array.isArray(allowed) || allowed.length === 0 || !allowed.every(entry => isStructuredScalar(entry))) {
|
||||
violations.push(`${path}.enum must be a non-empty array of scalars`)
|
||||
}
|
||||
}
|
||||
if ('const' in node && !isStructuredScalar(node.const)) {
|
||||
violations.push(`${path}.const must be a scalar`)
|
||||
}
|
||||
break
|
||||
}
|
||||
/* v8 ignore start -- defensive: schemaType was membership-checked against SCHEMA_TYPES above, so no runtime value reaches here */
|
||||
default:
|
||||
assertNever(schemaType, 'assertSupportedOutputSchema')
|
||||
/* v8 ignore stop */
|
||||
}
|
||||
|
||||
seen.delete(node)
|
||||
}
|
||||
|
||||
/**
|
||||
* Assert `schema` is a supported {@link StructuredOutputSchema} — object-rooted
|
||||
* and entirely within the enforced subset. Throws {@link OutputSchemaError}
|
||||
* (`UNSUPPORTED_SCHEMA`) listing EVERY violation; returns (and narrows) on
|
||||
* success. Call this at the seam boundary, before any child is created.
|
||||
* @param schema - the caller-supplied schema (unknown until asserted).
|
||||
* @returns nothing — the assertion signature narrows `schema` to
|
||||
* {@link StructuredOutputSchema} in the caller's scope on normal return.
|
||||
*/
|
||||
export function assertSupportedOutputSchema(schema: unknown): asserts schema is StructuredOutputSchema {
|
||||
const violations: string[] = []
|
||||
checkSchemaNode(schema, 'schema', violations, new Set())
|
||||
if (violations.length === 0 && (schema as StructuredSchemaNode).type !== 'object') {
|
||||
violations.push('schema.type must be "object" (structured output is object-rooted)')
|
||||
}
|
||||
if (violations.length > 0) throw new OutputSchemaError(violations)
|
||||
}
|
||||
|
||||
/** Collect violations for one value against an (already asserted) schema node. */
|
||||
function checkValue(node: StructuredSchemaNode, value: unknown, path: string): string[] {
|
||||
switch (node.type) {
|
||||
case 'object': {
|
||||
if (!isObjectLike(value)) return [`"${path}" must be an object`]
|
||||
const violations: string[] = []
|
||||
const properties = node.properties ?? {}
|
||||
// Own-property discipline throughout: JSON carries own enumerable
|
||||
// properties only, so an inherited `toString` must not satisfy
|
||||
// `required`, dodge `additionalProperties: false`, or be validated as if
|
||||
// the value carried it.
|
||||
for (const key of node.required ?? []) {
|
||||
if (!Object.hasOwn(value, key) || value[key] === undefined) violations.push(`missing required property "${path}.${key}"`)
|
||||
}
|
||||
for (const [key, child] of Object.entries(properties)) {
|
||||
if (!Object.hasOwn(value, key) || value[key] === undefined) continue
|
||||
violations.push(...checkValue(child, value[key], `${path}.${key}`))
|
||||
}
|
||||
if (node.additionalProperties === false) {
|
||||
for (const key of Object.keys(value)) {
|
||||
if (!Object.hasOwn(properties, key)) violations.push(`"${path}.${key}" is not a declared property (additionalProperties: false)`)
|
||||
}
|
||||
}
|
||||
return violations
|
||||
}
|
||||
case 'array': {
|
||||
if (!Array.isArray(value)) return [`"${path}" must be an array`]
|
||||
if (!node.items) return []
|
||||
const items = node.items
|
||||
return value.flatMap((entry, index) => checkValue(items, entry, `${path}[${index}]`))
|
||||
}
|
||||
case 'string': {
|
||||
if (typeof value !== 'string') return [`"${path}" must be a string`]
|
||||
break
|
||||
}
|
||||
case 'number': {
|
||||
if (typeof value !== 'number' || !Number.isFinite(value)) return [`"${path}" must be a finite number`]
|
||||
break
|
||||
}
|
||||
case 'integer': {
|
||||
if (typeof value !== 'number' || !Number.isInteger(value)) return [`"${path}" must be an integer`]
|
||||
break
|
||||
}
|
||||
case 'boolean': {
|
||||
if (typeof value !== 'boolean') return [`"${path}" must be a boolean`]
|
||||
break
|
||||
}
|
||||
case 'null': {
|
||||
if (value !== null) return [`"${path}" must be null`]
|
||||
break
|
||||
}
|
||||
default:
|
||||
return assertNever(node.type, 'validateStructuredValue')
|
||||
}
|
||||
// Scalar constraint checks, shared by every scalar branch above.
|
||||
if (node.enum && !node.enum.includes(value)) {
|
||||
return [`"${path}" must be one of ${JSON.stringify(node.enum)}`]
|
||||
}
|
||||
if ('const' in node && value !== node.const) {
|
||||
return [`"${path}" must be ${JSON.stringify(node.const)}`]
|
||||
}
|
||||
return []
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate a value against an (already {@link assertSupportedOutputSchema}-
|
||||
* asserted) schema. Returns human-readable, path-qualified violation messages
|
||||
* — empty means valid. Total: never throws, however malformed the value.
|
||||
* @param schema - the asserted schema to check against.
|
||||
* @param value - the candidate value (e.g. parsed tool-call arguments).
|
||||
* @returns every violation found, in walk order (empty = valid).
|
||||
*/
|
||||
export function validateStructuredValue(schema: StructuredOutputSchema, value: unknown): string[] {
|
||||
return checkValue(schema, value, 'value')
|
||||
}
|
||||
@@ -155,6 +155,9 @@ export interface JsonSchemaObject {
|
||||
* `properties`, `required` array).
|
||||
*
|
||||
* This is a plain function — no schemastery or other framework dependency.
|
||||
* @param spec - the author-facing per-property schema to convert.
|
||||
* @returns the wire-format JSON Schema; the top-level `required` array is
|
||||
* omitted entirely when no property is marked required.
|
||||
*/
|
||||
export function schemaSpecToJsonSchema(spec: SchemaSpec): JsonSchemaObject {
|
||||
const properties: Record<string, unknown> = {}
|
||||
@@ -269,6 +272,9 @@ function checkSpec(spec: SchemaSpec, value: unknown, path: string): string[] {
|
||||
* keys are allowed (no `additionalProperties: false`); `default` is not
|
||||
* applied; an `object`/`array` prop without `properties`/`items` only
|
||||
* type-checks; `enum` is membership (strings only).
|
||||
* @param spec - the declared parameter schema to validate against.
|
||||
* @param args - the model-generated arguments, however malformed.
|
||||
* @returns the violation messages in declaration order; empty means valid.
|
||||
*/
|
||||
export function validateArgs(spec: SchemaSpec, args: unknown): string[] {
|
||||
return checkSpec(spec, args, '')
|
||||
@@ -340,6 +346,13 @@ export interface DefineToolOptions<S extends SchemaSpec> {
|
||||
* Raw JSON-Schema tool definitions (from MCP servers) are still accepted
|
||||
* by `ToolRegistry.register()` directly — `defineTool` is sugar for
|
||||
* first-party plugin authors.
|
||||
* @param options - the tool's name, description, typed parameter schema,
|
||||
* execute body, and optional presenters.
|
||||
* @returns a registry-ready {@link ToolDefinition}: its `execute` validates the
|
||||
* raw args first (throwing {@link ToolArgsError} on mismatch, which the
|
||||
* registry turns into an isError result), and its presenters validate softly
|
||||
* (returning undefined on mismatch, since replay may feed them older-schema
|
||||
* args).
|
||||
*/
|
||||
export function defineTool<S extends SchemaSpec>(options: DefineToolOptions<S>): ToolDefinition {
|
||||
// Object-literal execute methods don't use `this`; the reference is safe.
|
||||
|
||||
304
packages/core/tools/tests/json-schema.spec.ts
Normal file
304
packages/core/tools/tests/json-schema.spec.ts
Normal file
@@ -0,0 +1,304 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import {
|
||||
assertSupportedOutputSchema,
|
||||
OutputSchemaError,
|
||||
validateStructuredValue,
|
||||
type StructuredOutputSchema,
|
||||
} from '../src/json-schema.ts'
|
||||
|
||||
/** Assert-and-narrow helper: the asserted schema, typed. */
|
||||
function asserted(schema: unknown): StructuredOutputSchema {
|
||||
assertSupportedOutputSchema(schema)
|
||||
return schema
|
||||
}
|
||||
|
||||
/** The violations OutputSchemaError carries for a bad schema (throws if it passes). */
|
||||
function violationsOf(schema: unknown): string[] {
|
||||
try {
|
||||
assertSupportedOutputSchema(schema)
|
||||
} catch (error: unknown) {
|
||||
if (error instanceof OutputSchemaError) return error.violations
|
||||
throw error
|
||||
}
|
||||
throw new Error('expected the schema to be rejected')
|
||||
}
|
||||
|
||||
describe('assertSupportedOutputSchema', () => {
|
||||
it('accepts a representative subset schema (all supported keywords)', () => {
|
||||
const schema = asserted({
|
||||
type: 'object',
|
||||
description: 'a finding',
|
||||
title: 'Finding',
|
||||
properties: {
|
||||
file: { type: 'string', description: 'path' },
|
||||
line: { type: 'integer' },
|
||||
severity: { type: 'string', enum: ['low', 'high'] },
|
||||
kind: { type: 'string', const: 'bug' },
|
||||
score: { type: 'number' },
|
||||
confirmed: { type: 'boolean' },
|
||||
parent: { type: 'null' },
|
||||
tags: { type: 'array', items: { type: 'string' } },
|
||||
nested: {
|
||||
type: 'object',
|
||||
properties: { x: { type: 'number', default: 3, examples: [1, 2] } },
|
||||
additionalProperties: false,
|
||||
},
|
||||
anything: { type: 'array' },
|
||||
},
|
||||
required: ['file', 'line'],
|
||||
additionalProperties: true,
|
||||
})
|
||||
expect(schema.type).toBe('object')
|
||||
})
|
||||
|
||||
it('rejects a non-object root (scalar/array-rooted schemas)', () => {
|
||||
expect(violationsOf({ type: 'string' })).toEqual(['schema.type must be "object" (structured output is object-rooted)'])
|
||||
expect(violationsOf({ type: 'array', items: { type: 'string' } }))
|
||||
.toContain('schema.type must be "object" (structured output is object-rooted)')
|
||||
})
|
||||
|
||||
it('rejects non-object schema nodes and missing/unknown type', () => {
|
||||
expect(violationsOf('nope')).toEqual(['schema must be a schema object'])
|
||||
expect(violationsOf(null)).toEqual(['schema must be a schema object'])
|
||||
expect(violationsOf([])).toEqual(['schema must be a schema object'])
|
||||
expect(violationsOf({})).toEqual(['schema.type must be one of object/array/string/number/integer/boolean/null'])
|
||||
expect(violationsOf({ type: 'tuple' })[0]).toMatch(/type must be one of/)
|
||||
expect(violationsOf({ type: 'object', properties: { a: 'str' } })).toEqual(['schema.properties.a must be a schema object'])
|
||||
})
|
||||
|
||||
it('rejects type ARRAYS with a dedicated message', () => {
|
||||
expect(violationsOf({ type: ['string', 'null'] }))
|
||||
.toEqual(['schema.type must be a single type string (type arrays are not supported)'])
|
||||
})
|
||||
|
||||
it('rejects unsupported constraint keywords loudly (never accepted-then-ignored)', () => {
|
||||
for (const keyword of ['oneOf', 'anyOf', 'allOf', 'not', 'pattern', 'minimum', 'maxLength', '$ref']) {
|
||||
const bad = violationsOf({ type: 'object', [keyword]: [] })
|
||||
expect(bad.some(v => v.includes(`schema.${keyword} is not a supported keyword`))).toBe(true)
|
||||
}
|
||||
})
|
||||
|
||||
it('reports EVERY violation, not just the first', () => {
|
||||
const bad = violationsOf({
|
||||
type: 'object',
|
||||
pattern: 'x',
|
||||
properties: { a: { type: 'weird' }, b: { type: 'string', minimum: 1 } },
|
||||
})
|
||||
expect(bad.length).toBe(3)
|
||||
})
|
||||
|
||||
it('rejects keywords on the wrong type (items on object, properties on string, enum on object)', () => {
|
||||
expect(violationsOf({ type: 'object', items: { type: 'string' } }))
|
||||
.toEqual(['schema.items is not supported on type "object"'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'string', properties: {} } } }))
|
||||
.toEqual(['schema.properties.a.properties is not supported on type "string"'])
|
||||
expect(violationsOf({ type: 'object', enum: [1] }))
|
||||
.toEqual(['schema.enum is not supported on type "object"'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'array', const: 1 } } }))
|
||||
.toEqual(['schema.properties.a.const is not supported on type "array"'])
|
||||
})
|
||||
|
||||
it('validates required: must be string[] naming declared properties', () => {
|
||||
expect(violationsOf({ type: 'object', required: 'file' }))
|
||||
.toEqual(['schema.required must be an array of strings'])
|
||||
expect(violationsOf({ type: 'object', required: [1] }))
|
||||
.toEqual(['schema.required must be an array of strings'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'string' } }, required: ['b'] }))
|
||||
.toEqual(['schema.required names "b" which is not in properties'])
|
||||
expect(violationsOf({ type: 'object', required: ['a'] }))
|
||||
.toEqual(['schema.required names "a" which is not in properties'])
|
||||
})
|
||||
|
||||
it('validates additionalProperties must be boolean and enum/const must be scalars', () => {
|
||||
expect(violationsOf({ type: 'object', additionalProperties: {} }))
|
||||
.toEqual(['schema.additionalProperties must be a boolean'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'string', enum: [] } } }))
|
||||
.toEqual(['schema.properties.a.enum must be a non-empty array of scalars'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'string', enum: [{}] } } }))
|
||||
.toEqual(['schema.properties.a.enum must be a non-empty array of scalars'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'string', enum: 'x' } } }))
|
||||
.toEqual(['schema.properties.a.enum must be a non-empty array of scalars'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'number', enum: [Number.NaN] } } }))
|
||||
.toEqual(['schema.properties.a.enum must be a non-empty array of scalars'])
|
||||
expect(violationsOf({ type: 'object', properties: { a: { type: 'string', const: {} } } }))
|
||||
.toEqual(['schema.properties.a.const must be a scalar'])
|
||||
})
|
||||
|
||||
it('rejects non-string description/title and non-JSON annotation payloads', () => {
|
||||
expect(violationsOf({ type: 'object', description: 7 }))
|
||||
.toEqual(['schema.description must be a string'])
|
||||
expect(violationsOf({ type: 'object', title: 7 }))
|
||||
.toEqual(['schema.title must be a string'])
|
||||
expect(violationsOf({ type: 'object', default: () => 1 }))
|
||||
.toEqual(['schema.default annotation must be JSON data'])
|
||||
expect(violationsOf({ type: 'object', examples: [undefined] }))
|
||||
.toEqual(['schema.examples annotation must be JSON data'])
|
||||
expect(violationsOf({ type: 'object', examples: [Number.POSITIVE_INFINITY] }))
|
||||
.toEqual(['schema.examples annotation must be JSON data'])
|
||||
// A cyclic annotation payload is caught by the JSON-data walk.
|
||||
const cyclicAnnotation: Record<string, unknown> = {}
|
||||
cyclicAnnotation.self = cyclicAnnotation
|
||||
expect(violationsOf({ type: 'object', default: cyclicAnnotation }))
|
||||
.toEqual(['schema.default annotation must be JSON data'])
|
||||
// Object/array annotations that ARE JSON data pass.
|
||||
asserted({ type: 'object', default: { a: [1, 'x', null, true] } })
|
||||
})
|
||||
|
||||
it('rejects a circular schema instead of recursing forever', () => {
|
||||
const node: Record<string, unknown> = { type: 'object' }
|
||||
node.properties = { self: node }
|
||||
expect(violationsOf(node)).toEqual(['schema.properties.self is circular'])
|
||||
})
|
||||
|
||||
it('accepts the same subschema object reused in two SIBLING positions (a DAG, not a cycle)', () => {
|
||||
const leaf = { type: 'string' }
|
||||
asserted({ type: 'object', properties: { a: leaf, b: leaf } })
|
||||
})
|
||||
|
||||
it('required cannot be satisfied by INHERITED names — `toString` is not a declared property', () => {
|
||||
// `'toString' in {}` is true via Object.prototype; the declared-property
|
||||
// contract must be an own-property check.
|
||||
expect(violationsOf({ type: 'object', properties: {}, required: ['toString'] }))
|
||||
.toEqual(['schema.required names "toString" which is not in properties'])
|
||||
})
|
||||
|
||||
it('rejects exotic host objects where the subset expects plain JSON structure', () => {
|
||||
// A Map as `properties` has no own enumerable entries: structurally it
|
||||
// would read as "no properties" and serialize to {} — lossy, not loud.
|
||||
expect(violationsOf({ type: 'object', properties: new Map() }))
|
||||
.toEqual(['schema.properties must be an object of schemas'])
|
||||
// A Date node is not a schema object even though Object.values(date) is [].
|
||||
expect(violationsOf({ type: 'object', properties: { at: new Date(0) } }))
|
||||
.toEqual(['schema.properties.at must be a schema object'])
|
||||
})
|
||||
|
||||
it('rejects exotic annotation payloads that would serialize lossily', () => {
|
||||
expect(violationsOf({ type: 'object', default: new Date(0) }))
|
||||
.toEqual(['schema.default annotation must be JSON data'])
|
||||
expect(violationsOf({ type: 'object', examples: [new Map()] }))
|
||||
.toEqual(['schema.examples annotation must be JSON data'])
|
||||
})
|
||||
})
|
||||
|
||||
describe('validateStructuredValue', () => {
|
||||
const schema = asserted({
|
||||
type: 'object',
|
||||
properties: {
|
||||
file: { type: 'string' },
|
||||
line: { type: 'integer' },
|
||||
score: { type: 'number' },
|
||||
confirmed: { type: 'boolean' },
|
||||
parent: { type: 'null' },
|
||||
severity: { type: 'string', enum: ['low', 'high'] },
|
||||
kind: { type: 'string', const: 'bug' },
|
||||
tags: { type: 'array', items: { type: 'string' } },
|
||||
free: { type: 'array' },
|
||||
nested: { type: 'object', properties: { x: { type: 'number' } }, required: ['x'], additionalProperties: false },
|
||||
},
|
||||
required: ['file'],
|
||||
})
|
||||
|
||||
it('accepts a fully valid value (empty violations)', () => {
|
||||
expect(validateStructuredValue(schema, {
|
||||
file: 'a.ts', line: 3, score: 0.5, confirmed: true, parent: null,
|
||||
severity: 'high', kind: 'bug', tags: ['x'], free: [1, { any: true }], nested: { x: 1 },
|
||||
})).toEqual([])
|
||||
})
|
||||
|
||||
it('reports missing required and wrong root type', () => {
|
||||
expect(validateStructuredValue(schema, {})).toEqual(['missing required property "value.file"'])
|
||||
expect(validateStructuredValue(schema, 'nope')).toEqual(['"value" must be an object'])
|
||||
expect(validateStructuredValue(schema, [])).toEqual(['"value" must be an object'])
|
||||
})
|
||||
|
||||
it('type-checks every scalar branch with path-qualified messages', () => {
|
||||
expect(validateStructuredValue(schema, { file: 1 })).toEqual(['"value.file" must be a string'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', line: 1.5 })).toEqual(['"value.line" must be an integer'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', line: 'x' })).toEqual(['"value.line" must be an integer'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', score: 'x' })).toEqual(['"value.score" must be a finite number'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', score: Number.NaN })).toEqual(['"value.score" must be a finite number'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', confirmed: 'yes' })).toEqual(['"value.confirmed" must be a boolean'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', parent: 0 })).toEqual(['"value.parent" must be null'])
|
||||
})
|
||||
|
||||
it('enforces enum membership and const equality', () => {
|
||||
expect(validateStructuredValue(schema, { file: 'a', severity: 'mid' }))
|
||||
.toEqual(['"value.severity" must be one of ["low","high"]'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', kind: 'feature' }))
|
||||
.toEqual(['"value.kind" must be "bug"'])
|
||||
})
|
||||
|
||||
it('checks arrays per index; an items-less array accepts anything', () => {
|
||||
expect(validateStructuredValue(schema, { file: 'a', tags: 'x' })).toEqual(['"value.tags" must be an array'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', tags: ['ok', 2] })).toEqual(['"value.tags[1]" must be a string'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', free: [{ deep: [1] }, null] })).toEqual([])
|
||||
})
|
||||
|
||||
it('recurses into nested objects: required + additionalProperties: false', () => {
|
||||
expect(validateStructuredValue(schema, { file: 'a', nested: {} }))
|
||||
.toEqual(['missing required property "value.nested.x"'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', nested: { x: 1, y: 2 } }))
|
||||
.toEqual(['"value.nested.y" is not a declared property (additionalProperties: false)'])
|
||||
expect(validateStructuredValue(schema, { file: 'a', nested: 3 }))
|
||||
.toEqual(['"value.nested" must be an object'])
|
||||
})
|
||||
|
||||
it('a required key present-but-undefined counts as missing', () => {
|
||||
expect(validateStructuredValue(schema, { file: undefined })).toEqual(['missing required property "value.file"'])
|
||||
})
|
||||
|
||||
it('inherited properties satisfy nothing: required, additionalProperties, and recursion are own-property only', () => {
|
||||
// required: ['toString'] must NOT be satisfied by Object.prototype.toString.
|
||||
expect(validateStructuredValue(
|
||||
asserted({ type: 'object', properties: { toString: { type: 'string' } }, required: ['toString'] }),
|
||||
{},
|
||||
)).toEqual(['missing required property "value.toString"'])
|
||||
// additionalProperties: false must flag an OWN `toString` key even though
|
||||
// `'toString' in properties` is true via the prototype.
|
||||
expect(validateStructuredValue(
|
||||
asserted({ type: 'object', additionalProperties: false }),
|
||||
{ toString: 1 },
|
||||
)).toEqual(['"value.toString" is not a declared property (additionalProperties: false)'])
|
||||
// A declared property the value does NOT carry must not be validated
|
||||
// against the value's INHERITED member (constructor is a function on
|
||||
// every plain object's prototype, not a carried property).
|
||||
expect(validateStructuredValue(
|
||||
asserted({ type: 'object', properties: { constructor: { type: 'string' } } }),
|
||||
{},
|
||||
)).toEqual([])
|
||||
})
|
||||
|
||||
it('a non-plain object value is not an object in the JSON sense', () => {
|
||||
expect(validateStructuredValue(asserted({ type: 'object' }), new Date(0)))
|
||||
.toEqual(['"value" must be an object'])
|
||||
})
|
||||
|
||||
it('collects multiple violations across branches in one pass', () => {
|
||||
expect(validateStructuredValue(schema, { line: 'x', severity: 'mid' })).toEqual([
|
||||
'missing required property "value.file"',
|
||||
'"value.line" must be an integer',
|
||||
'"value.severity" must be one of ["low","high"]',
|
||||
])
|
||||
})
|
||||
|
||||
it('null-typed const/enum work through the scalar path', () => {
|
||||
const nullish = asserted({ type: 'object', properties: { a: { type: 'null', const: null } } })
|
||||
expect(validateStructuredValue(nullish, { a: null })).toEqual([])
|
||||
})
|
||||
|
||||
it('rejects a non-object properties value in the schema walk', () => {
|
||||
expect(violationsOf({ type: 'object', properties: [] }))
|
||||
.toEqual(['schema.properties must be an object of schemas'])
|
||||
})
|
||||
|
||||
it('an object schema without properties/required only type-checks its value', () => {
|
||||
const bare = asserted({ type: 'object' })
|
||||
expect(validateStructuredValue(bare, { any: ['thing'] })).toEqual([])
|
||||
expect(validateStructuredValue(bare, 7)).toEqual(['"value" must be an object'])
|
||||
})
|
||||
|
||||
it('validateStructuredValue throws on a type the assert would never let through (assertNever backstop)', () => {
|
||||
const forged = { type: 'tuple' } as unknown as StructuredOutputSchema
|
||||
expect(() => validateStructuredValue(forged, 1)).toThrow(/tuple/)
|
||||
})
|
||||
})
|
||||
@@ -129,6 +129,9 @@ export interface LocalDirEntry {
|
||||
* and intermediate directories are created by the write. Two input paths
|
||||
* reaching the same file via symlinks share one key. Falls back to the absolute
|
||||
* path only when no ancestor (not even the filesystem root) can be resolved.
|
||||
* @param cwd - base directory a relative `path` resolves against.
|
||||
* @param path - absolute or relative path; empty/whitespace-only throws `FS_NOT_FOUND`.
|
||||
* @returns the absolute display path plus the realpath-derived stable target key.
|
||||
*/
|
||||
export async function resolveLocalTarget(cwd: string, path: string): Promise<LocalTarget> {
|
||||
if (path.trim().length === 0) throw new FsError('file_path must be a non-empty string', 'FS_NOT_FOUND')
|
||||
@@ -165,7 +168,11 @@ export async function resolveLocalTarget(cwd: string, path: string): Promise<Loc
|
||||
}
|
||||
}
|
||||
|
||||
/** Probe a path for its version, mode, type, and size. Null if absent. */
|
||||
/**
|
||||
* Probe a path for its version, mode, type, and size. Null if absent.
|
||||
* @param absolutePath - the path to stat (typically a target key; symlinks are followed).
|
||||
* @returns the metadata, or null when the path — or a parent segment — does not exist.
|
||||
*/
|
||||
export async function probe(absolutePath: string): Promise<PathInfo | null> {
|
||||
try {
|
||||
const info = await stat(absolutePath)
|
||||
@@ -200,6 +207,9 @@ async function resolveListedChildTarget(parent: LocalTarget, name: string): Prom
|
||||
* List direct children of a directory in stable name order. Each child includes
|
||||
* a resolved target plus stat metadata when still available; file contents are
|
||||
* never read.
|
||||
* @param target - the resolved directory to list; a missing or non-directory target throws.
|
||||
* @param signal - aborts the listing, checked between children (`FS_ABORTED`).
|
||||
* @returns one entry per direct child, sorted by name.
|
||||
*/
|
||||
export async function listDirectory(target: LocalTarget, signal?: AbortSignal): Promise<LocalDirEntry[]> {
|
||||
throwIfAborted(signal, 'list')
|
||||
@@ -290,6 +300,9 @@ async function statRegularFile(target: LocalTarget, verb: 'read', signal?: Abort
|
||||
/**
|
||||
* Read a whole regular UTF-8 text file into a single decoded string. Rejects
|
||||
* non-regular files, invalid UTF-8, and NUL-byte binary samples.
|
||||
* @param target - the resolved file to read.
|
||||
* @param signal - aborts the read (`FS_ABORTED`).
|
||||
* @returns the full decoded text, byte-for-byte (no normalization).
|
||||
*/
|
||||
export async function readWholeText(target: LocalTarget, signal?: AbortSignal): Promise<string> {
|
||||
await statRegularFile(target, 'read', signal)
|
||||
@@ -305,6 +318,9 @@ export async function readWholeText(target: LocalTarget, signal?: AbortSignal):
|
||||
* Stream a whole regular UTF-8 text file as decoded text chunks. Same text
|
||||
* semantics as {@link readWholeText} (regular-file check, binary/NUL rejection,
|
||||
* cross-chunk UTF-8 decoding), but never holds the whole file in memory.
|
||||
* @param target - the resolved file to stream.
|
||||
* @param signal - aborts the stream, including between chunks (`FS_ABORTED`).
|
||||
* @returns decoded text chunks in file order; chunk boundaries carry no meaning.
|
||||
*/
|
||||
export async function* streamWholeText(target: LocalTarget, signal?: AbortSignal): AsyncIterable<string> {
|
||||
await statRegularFile(target, 'read', signal)
|
||||
@@ -352,6 +368,11 @@ async function removeStagingDirOrThrow(stagingDir: string, originalError: unknow
|
||||
* (`0o700`) staging directory, fsync, optionally chmod to the final mode while
|
||||
* still private, then rename over the target. `mode` (when given) preserves an
|
||||
* existing file's permissions across the replace.
|
||||
* @param absolutePath - the final destination (typically a target key); missing parent dirs are created.
|
||||
* @param content - the full UTF-8 text to write.
|
||||
* @param mode - final file mode applied before the rename (an existing file's, to preserve permissions); undefined leaves `0o600`.
|
||||
* @param signal - aborts the write (`FS_ABORTED`); checked before the rename, so the target is never left torn.
|
||||
* @param internals - test seam for pinning temp names and observing the staged file.
|
||||
*/
|
||||
export async function writeFileAtomic(
|
||||
absolutePath: string,
|
||||
@@ -409,6 +430,12 @@ export async function writeFileAtomic(
|
||||
/** Line ending style detected before LF normalization. */
|
||||
export type LineEndings = 'LF' | 'CRLF'
|
||||
|
||||
/**
|
||||
* Collapse CRLF to LF — the canonical in-memory form every edit/diff basis
|
||||
* uses. Lone `\r` bytes (not followed by `\n`) are left untouched.
|
||||
* @param content - decoded text in whatever line-ending style the file had.
|
||||
* @returns the text with every `\r\n` pair replaced by `\n`.
|
||||
*/
|
||||
function normalizeLineEndings(content: string): string {
|
||||
return content.replaceAll('\r\n', '\n')
|
||||
}
|
||||
@@ -420,6 +447,14 @@ function detectLineEndings(raw: string): LineEndings {
|
||||
return crlfCount > lfCount ? 'CRLF' : 'LF'
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert LF-normalized content back to the line-ending style detected at read
|
||||
* time, for write-back. `LF` returns the content unchanged; `CRLF` re-normalizes
|
||||
* first so an already-CRLF sequence is never doubled to `\r\r\n`.
|
||||
* @param content - the LF-normalized (edited) text.
|
||||
* @param lineEndings - the original file's style, as detected by {@link readForEdit}.
|
||||
* @returns the text in the original file's line-ending style.
|
||||
*/
|
||||
function restoreLineEndings(content: string, lineEndings: LineEndings): string {
|
||||
return lineEndings === 'LF' ? content : normalizeLineEndings(content).split('\n').join('\r\n')
|
||||
}
|
||||
@@ -438,6 +473,10 @@ function countOccurrences(content: string, needle: string): number {
|
||||
/**
|
||||
* Read and decode a file for editing: rejects binaries, returns LF-normalized
|
||||
* content plus the original line-ending style for write-back.
|
||||
* @param absolutePath - the file to read (typically a target key).
|
||||
* @param displayPath - the caller-facing path used in error messages.
|
||||
* @param signal - aborts the read (`FS_ABORTED`).
|
||||
* @returns the LF-normalized content and the detected style to restore on write-back.
|
||||
*/
|
||||
export async function readForEdit(
|
||||
absolutePath: string,
|
||||
@@ -459,6 +498,9 @@ export async function readForEdit(
|
||||
* prior bytes, so an undiffable prior file simply yields no contextual-hunk basis
|
||||
* (the caller treats `null` the same as an absent file: the result renders a
|
||||
* whole-file diff rather than an applied hunk).
|
||||
* @param absolutePath - the file to read (typically a target key); it must exist.
|
||||
* @param signal - aborts the read (`FS_ABORTED`).
|
||||
* @returns the LF-normalized text, or null for a binary or non-UTF-8 file.
|
||||
*/
|
||||
export async function readTextForDiff(absolutePath: string, signal?: AbortSignal): Promise<string | null> {
|
||||
const buffer = await readFileAbortable(absolutePath, 'read', signal)
|
||||
@@ -477,6 +519,12 @@ export async function readTextForDiff(absolutePath: string, signal?: AbortSignal
|
||||
* `FS_EDIT_NOT_FOUND` on empty `oldString` or zero matches and
|
||||
* `FS_AMBIGUOUS_EDIT` on multiple matches when `replaceAll` is false. Returns
|
||||
* the edited content (still LF-normalized) and the replacement count.
|
||||
* @param content - the current file content, already LF-normalized.
|
||||
* @param oldString - literal text to find; CRLF inside it is normalized to LF before matching.
|
||||
* @param newString - literal replacement text, normalized the same way.
|
||||
* @param replaceAll - replace every match instead of requiring exactly one.
|
||||
* @param displayPath - the caller-facing path used in error messages.
|
||||
* @returns the edited LF-normalized content plus how many occurrences were replaced.
|
||||
*/
|
||||
export function applyLiteralEdit(
|
||||
content: string,
|
||||
|
||||
@@ -73,6 +73,7 @@ export class LocalFileSystem extends FileSystem {
|
||||
cwd: z.string().default(process.cwd()),
|
||||
})
|
||||
|
||||
/** Validated config (schemastery applied the defaults before construction). */
|
||||
readonly config: ResolvedConfig
|
||||
/** Test seam forwarded to fsio (force streaming path, pin temp names). */
|
||||
internals: FsIoInternals = {}
|
||||
|
||||
@@ -29,7 +29,12 @@ import type { Branded } from '@deepseek-ai/dsh-brand'
|
||||
*/
|
||||
export type FsTargetKey = Branded<'FsTargetKey'>
|
||||
|
||||
/** Brand a string as an {@link FsTargetKey}. */
|
||||
/**
|
||||
* Brand a string as an {@link FsTargetKey}. For backend use only — a consumer
|
||||
* never manufactures a key, it receives one from `resolve()`.
|
||||
* @param key - the backend's raw key string (the local backend passes a realpath).
|
||||
* @returns the same string, branded; no validation is performed.
|
||||
*/
|
||||
export function FsTargetKey(key: string): FsTargetKey {
|
||||
return key as FsTargetKey
|
||||
}
|
||||
@@ -42,7 +47,12 @@ export function FsTargetKey(key: string): FsTargetKey {
|
||||
*/
|
||||
export type FsVersion = Branded<'FsVersion'>
|
||||
|
||||
/** Brand a string as an {@link FsVersion}. */
|
||||
/**
|
||||
* Brand a string as an {@link FsVersion}. For backend use only — a consumer
|
||||
* never manufactures a version, it receives one from `stat`/write/edit outcomes.
|
||||
* @param v - the backend's raw version string (the local backend derives it from mtime+size).
|
||||
* @returns the same string, branded; no validation is performed.
|
||||
*/
|
||||
export function FsVersion(v: string): FsVersion {
|
||||
return v as FsVersion
|
||||
}
|
||||
|
||||
@@ -40,6 +40,10 @@ export type FsDiffMeta = { diffs: FileDiff[] }
|
||||
* (a pure insertion) reports `oldText: null` (nothing to diff against), mirroring
|
||||
* the call-time card's new-file convention. The unified-diff "\ No newline at end
|
||||
* of file" markers are dropped — they annotate the patch, not file content.
|
||||
* @param path - the path stamped on every produced diff (the model-facing `file_path`; the bridge relativizes it).
|
||||
* @param before - the file text before the change (the backend's LF-normalized diff basis).
|
||||
* @param after - the file text after the change, on the same basis.
|
||||
* @returns one diff per applied hunk, in file order; empty when the texts are identical.
|
||||
*/
|
||||
export function computeHunkDiffs(path: string, before: string, after: string): FileDiff[] {
|
||||
const patch = structuredPatch('', '', before, after, undefined, undefined, { context: DIFF_CONTEXT })
|
||||
@@ -83,6 +87,8 @@ function isFileDiff(value: unknown): value is FileDiff {
|
||||
* it validates defensively rather than trusting the payload — a bad `meta` yields
|
||||
* `undefined`, and the caller decides the fallback (edit → the generic result
|
||||
* rendering; write → an args-derived whole-file diff), never a thrown presenter.
|
||||
* @param meta - the opaque `tool/result` meta payload (live or replayed from the session log).
|
||||
* @returns the validated non-empty hunk list, or undefined for an absent/empty/malformed payload.
|
||||
*/
|
||||
export function diffsFromMeta(meta: unknown): FileDiff[] | undefined {
|
||||
if (typeof meta !== 'object' || meta === null || Array.isArray(meta)) return undefined
|
||||
|
||||
@@ -29,7 +29,13 @@ interface EditInput {
|
||||
replaceAll: boolean
|
||||
}
|
||||
|
||||
/** Validate value constraints the schema DSL can't express. */
|
||||
/**
|
||||
* Validate value constraints the schema DSL can't express: a non-blank
|
||||
* `file_path`, a non-empty `old_string`, and `old_string !== new_string`
|
||||
* (an equal pair would be a guaranteed no-op edit).
|
||||
* @param args - the schema-validated raw tool arguments.
|
||||
* @returns the camelCased input with `replace_all` defaulted to false.
|
||||
*/
|
||||
export function parseEditArgs(args: { file_path: string; old_string: string; new_string: string; replace_all?: boolean }): EditInput {
|
||||
if (args.file_path.trim().length === 0) throw new Error('file_path must be a non-empty string')
|
||||
if (args.old_string.length === 0) throw new Error('old_string must be a non-empty string')
|
||||
@@ -42,14 +48,22 @@ export function parseEditArgs(args: { file_path: string; old_string: string; new
|
||||
}
|
||||
}
|
||||
|
||||
/** Format an edit success (single-match or replace-all) as a Claude-style model-facing message. */
|
||||
/**
|
||||
* Format an edit success (single-match or replace-all) as a Claude-style model-facing message.
|
||||
* @param displayPath - the backend-resolved path shown to the model.
|
||||
* @param replaceAll - selects the all-occurrences wording over the single-replacement one.
|
||||
* @returns the confirmation sentence the model sees as the tool result.
|
||||
*/
|
||||
export function formatEditOutput(displayPath: string, replaceAll: boolean): string {
|
||||
return replaceAll
|
||||
? `The file ${displayPath} has been updated. All occurrences were successfully replaced.`
|
||||
: `The file ${displayPath} has been updated successfully.`
|
||||
}
|
||||
|
||||
/** Register the `edit` tool and its system-prompt guidance. */
|
||||
/**
|
||||
* Register the `edit` tool and its system-prompt guidance.
|
||||
* @param ctx - the plugin context; registrations are effects scoped to it, and execution uses its `fs` service.
|
||||
*/
|
||||
export function applyEditTool(ctx: Context): void {
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tool:edit',
|
||||
|
||||
@@ -119,6 +119,10 @@ function finish(acc: WindowAccumulator, request: ReadWindow, displayPath: string
|
||||
* path serves both. Scans for newlines with a capped line buffer (a newline-free
|
||||
* giant line is truncated, never buffered past `request.maxLineLength`),
|
||||
* enforces the byte cap, and throws `FS_NOT_FOUND` for an offset past EOF.
|
||||
* @param chunks - decoded text chunks in file order; chunk boundaries carry no meaning.
|
||||
* @param request - the resolved window; the caller has already applied its defaults and caps.
|
||||
* @param displayPath - the caller-facing path used in the offset-out-of-range error.
|
||||
* @returns the numbered window lines, the total line count seen, and the byte-cap truncation flag.
|
||||
*/
|
||||
export async function buildWindow(
|
||||
chunks: AsyncIterable<string> | Iterable<string>,
|
||||
@@ -156,7 +160,12 @@ export async function buildWindow(
|
||||
return finish(acc, request, displayPath)
|
||||
}
|
||||
|
||||
/** Format a read outcome as one OpenCode-style line-numbered text block body. */
|
||||
/**
|
||||
* Format a read outcome as one OpenCode-style line-numbered text block body.
|
||||
* @param displayPath - the backend-resolved path rendered in the envelope's `<path>` element.
|
||||
* @param outcome - the windowed read to render.
|
||||
* @returns the model-facing envelope: numbered lines plus a continuation or end-of-file footer.
|
||||
*/
|
||||
export function formatReadOutput(displayPath: string, outcome: FileReadOutcome): string {
|
||||
const endLine = outcome.lines.at(-1)?.number ?? Math.max(0, outcome.offset - 1)
|
||||
let footer: string
|
||||
|
||||
@@ -58,7 +58,12 @@ function parsePositiveInteger(value: number, name: string): number {
|
||||
return value
|
||||
}
|
||||
|
||||
/** Validate value constraints the schema DSL can't express. `maxLimit` is the deployment's line cap. */
|
||||
/**
|
||||
* Validate value constraints the schema DSL can't express. `maxLimit` is the deployment's line cap.
|
||||
* @param args - the schema-validated raw tool arguments; `offset`/`limit` must be positive integers when given.
|
||||
* @param maxLimit - the configured line cap: both the default `limit` and the largest one accepted.
|
||||
* @returns the validated input with `offset` defaulted to 1 and `limit` to `maxLimit`.
|
||||
*/
|
||||
export function parseReadArgs(args: { file_path: string; offset?: number; limit?: number }, maxLimit: number): ReadInput {
|
||||
if (args.file_path.trim().length === 0) throw new Error('file_path must be a non-empty string')
|
||||
const offset = args.offset === undefined ? 1 : parsePositiveInteger(args.offset, 'offset')
|
||||
@@ -67,7 +72,11 @@ export function parseReadArgs(args: { file_path: string; offset?: number; limit?
|
||||
return { filePath: args.file_path, offset, limit }
|
||||
}
|
||||
|
||||
/** Register the `read` tool and its system-prompt guidance. */
|
||||
/**
|
||||
* Register the `read` tool and its system-prompt guidance.
|
||||
* @param ctx - the plugin context; registrations are effects scoped to it, and execution uses its `fs` service.
|
||||
* @param caps - the deployment's resolved read caps (plugin config after defaulting).
|
||||
*/
|
||||
export function applyReadTool(ctx: Context, caps: ReadToolCaps): void {
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tool:read',
|
||||
|
||||
@@ -18,7 +18,11 @@
|
||||
|
||||
import type { ToolExecution } from '@deepseek-ai/dsh-tools'
|
||||
|
||||
/** The session workspace cwd for this call, or `undefined` when none applies. */
|
||||
/**
|
||||
* The session workspace cwd for this call, or `undefined` when none applies.
|
||||
* @param exec - the tool-execution context; only its optional `agent` is read.
|
||||
* @returns the calling agent's session cwd, or undefined for a non-agent caller (the backend then applies its own default).
|
||||
*/
|
||||
export function sessionCwd(exec: ToolExecution): string | undefined {
|
||||
return exec.agent?.session.header.cwd
|
||||
}
|
||||
|
||||
@@ -21,13 +21,23 @@ import type {} from '@deepseek-ai/dsh-system-prompt'
|
||||
import { computeHunkDiffs, diffsFromMeta, type FsDiffMeta } from './diff.ts'
|
||||
import { sessionCwd } from './session-cwd.ts'
|
||||
|
||||
/** Validate value constraints the schema DSL can't express. */
|
||||
/**
|
||||
* Validate value constraints the schema DSL can't express: only a non-blank
|
||||
* `file_path` — an empty `content` is legitimate (it writes an empty file).
|
||||
* @param args - the schema-validated raw tool arguments.
|
||||
* @returns the camelCased input; `content` passes through untouched.
|
||||
*/
|
||||
export function parseWriteArgs(args: { file_path: string; content: string }): { filePath: string; content: string } {
|
||||
if (args.file_path.trim().length === 0) throw new Error('file_path must be a non-empty string')
|
||||
return { filePath: args.file_path, content: args.content }
|
||||
}
|
||||
|
||||
/** Format a write outcome as one model-facing text block body. */
|
||||
/**
|
||||
* Format a write outcome as one model-facing text block body.
|
||||
* @param displayPath - the backend-resolved path rendered in the envelope's `<path>` element.
|
||||
* @param outcome - the write outcome; its `operation` selects the Created/Updated wording.
|
||||
* @returns the model-facing confirmation envelope (no file content is echoed back).
|
||||
*/
|
||||
export function formatWriteOutput(displayPath: string, outcome: FsWriteOutcome): string {
|
||||
const verb = outcome.operation === 'create' ? 'Created' : 'Updated'
|
||||
return `<path>${displayPath}</path>
|
||||
@@ -37,7 +47,10 @@ ${verb} file
|
||||
</content>`
|
||||
}
|
||||
|
||||
/** Register the `write` tool and its system-prompt guidance. */
|
||||
/**
|
||||
* Register the `write` tool and its system-prompt guidance.
|
||||
* @param ctx - the plugin context; registrations are effects scoped to it, and execution uses its `fs` service.
|
||||
*/
|
||||
export function applyWriteTool(ctx: Context): void {
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tool:write',
|
||||
|
||||
@@ -4,7 +4,7 @@ The hooks subsystem lets users extend the agent at lifecycle points the way Clau
|
||||
|
||||
| Package | Role | Shape |
|
||||
|---|---|---|
|
||||
| `hook-protocol/` | Shared wire-protocol core: matcher primitive, exit-code/stdout codec, `runHook` (via `ctx.bash`), most-restrictive merge, `hook/*` session events | library (no plugin) |
|
||||
| `hook-protocol/` | Shared wire-protocol core: matcher primitive, exit-code/stdout codec, `runHook` (via `ctx.bash`), most-restrictive merge, `hook/*` session events, detached-run quiescence | library (no plugin) |
|
||||
| `hooks-claude/` | Bridge for a Claude Code `hooks.json` / settings | plugin |
|
||||
| `hooks-codex/` | Bridge for a Codex `hooks.json` | plugin |
|
||||
|
||||
|
||||
@@ -13,6 +13,7 @@ Why a shared lib at all: Codex deliberately reimplements a *subset* of the Claud
|
||||
| Decode output | `parseHookOutput(exit, stdout, stderr)` → neutral `HookOutput` | maps the neutral `HookOutput` onto a seam-specific typed Decision |
|
||||
| Merge N hooks | `mergeHookOutputs(outputs)` → most-restrictive `MergedHookOutcome` | — |
|
||||
| Durable record | `appendHookInvoked` / `appendHookResult` (`hook/*` session events; the result's `decision`/`stderrSummary` derive from the `HookOutput` here) | calls them around each invocation |
|
||||
| Detached-run quiescence | `createDetachedRuns()` — track fire-and-forget run chains; `drain()` aborts, then awaits them | passes `signal` to each detached `runHook`, registers `drain` as its effect disposer |
|
||||
|
||||
## Primitives
|
||||
|
||||
@@ -20,10 +21,11 @@ Why a shared lib at all: Codex deliberately reimplements a *subset* of the Claud
|
||||
- **`runHook(bash, hook, options, now)`** — serialize `options.payload` to the hook's stdin (with a trailing newline iff `options.trailingNewline`), merge `options.env` after the executor's credential scrub (the `dsh-bash` trusted-plugin surface), honor the hook's `timeoutSec` (else `options.defaultTimeoutMs` — the bridge owns the default, its config defaulting to the lib's `DEFAULT_HOOK_TIMEOUT_MS` 10-minute reference), and decode the result (threading `options.expectedEventName` to the codec). Never throws: an executor rejection (infra fault) becomes a `HookOutput` with `exitCode: undefined` (a non-blocking error). `now` is injected for testable durations.
|
||||
- **`parseHookOutput(exitCode, stdout, stderr, expectedEventName?)`** — the exit-code + structured-stdout codec. Exit `0` → parse JSON stdout (lenient: non-JSON is left for the bridge); exit `2` → blocking error, `stderr` is the block reason (surfaced as `decision: 'block'`); other → non-blocking error. `hookSpecificOutput.permissionDecision` (allow/deny/ask) overrides a legacy top-level `decision`; `additionalContext`/`updatedInput`/`systemMessage`/`continue`/`stopReason` are parsed too. The schemas key the `hookSpecificOutput` block by `hookEventName`, so passing `expectedEventName` (the firing event) DISCARDS a block whose `hookEventName` names a different event — or omits it entirely — its event-scoped fields don't take effect (a `PreToolUse` block on a `Stop` hook is malformed, and so is a discriminator-less block that would otherwise apply to any event), while the event-agnostic top-level fields still apply. Pure and total.
|
||||
- **`mergeHookOutputs(outputs)`** — fold the results of every hook that matched one point: permission precedence **deny > ask > allow**, halt sticky on the first `continue:false`, block reasons joined with `\n\n`, `additionalContext`/`systemMessages` accumulated in order.
|
||||
- **`createDetachedRuns()`** — quiescence tracking for the emit-shaped points, which run detached (no seam awaits them). The bridge tracks each run chain — the hook run PLUS its continuation — and registers `drain()` as its effect disposer: drain fires the tracker's abort `signal` (so a still-running hook process is killed via `runHook`, not awaited out to its timeout), then resolves once every tracked chain has settled. `fiber.dispose()` resolving therefore means no detached hook work is left to fire into a disposed context ([defensive patterns](../../../docs/defensive-patterns.md): dispose must reach quiescence).
|
||||
|
||||
## `hook/*` session events
|
||||
|
||||
Declaration-merged into `SessionEventMap` (log-only, like `compact/*` — NOT a `SurfaceEventType`, no `surfaceOp`): `hook/invoked` (a hook command ran) and `hook/result` (its outcome, paired by `handlerId`, with `appendHookResult` owning the decision rule). Payloads and per-event JSDoc are in the generated [persistence log event catalog](../../../docs/persistence-catalog/log-events.md); `stderrSummary` is truncated to the record's `stderrSummaryMaxChars` (the bridge's config, reference default `DEFAULT_STDERR_SUMMARY_MAX_CHARS` = 500; omitted when empty).
|
||||
Declaration-merged into `SessionEventMap` (log-only, like `compact/*` — NOT a `SurfaceEventType`, no `surfaceOp`): `hook/invoked` (a hook command ran) and `hook/result` (its outcome, paired by `handlerId`, with `appendHookResult` owning the decision rule). Payloads and per-event JSDoc are in the generated [persistence log event catalog](../../../docs/persistence-catalog.md); `stderrSummary` is truncated to the record's `stderrSummaryMaxChars` (the bridge's config, reference default `DEFAULT_STDERR_SUMMARY_MAX_CHARS` = 500; omitted when empty).
|
||||
|
||||
Like every event they must sit inside an open turn. The mid-turn points (`PreToolUse`/`PostToolUse`/`UserPromptSubmit`/`Stop`) fire inside the loop's open turn by construction; `SessionStart` gets no `hook/*` record (its injected `context/message` is the durable evidence) — see the hooks RFC.
|
||||
|
||||
|
||||
@@ -75,6 +75,12 @@ function permissionDecisionOf(value: string | undefined): HookOutput['decision']
|
||||
* (`decision`/`reason`/`continue`/`stopReason`/`systemMessage`)
|
||||
* are unaffected. Omit `expectedEventName` (or pass a matching one) to apply the
|
||||
* block as-is — a caller that doesn't key by event opts out of the check.
|
||||
*
|
||||
* @param exitCode - the process exit code; `undefined` when the hook could not be spawned at all.
|
||||
* @param stdout - the captured stdout stream; consulted for structured JSON only on a 0 exit.
|
||||
* @param stderr - the captured stderr stream; becomes the blocking `reason` on exit 2.
|
||||
* @param expectedEventName - the event the hook is firing for; omit to apply a `hookSpecificOutput` block as-is.
|
||||
* @returns the dialect-neutral decoded outcome.
|
||||
*/
|
||||
export function parseHookOutput(exitCode: number | undefined, stdout: string, stderr: string, expectedEventName?: string): HookOutput {
|
||||
const trimmedErr = stderr.trim()
|
||||
|
||||
71
packages/hooks/hook-protocol/src/detached.ts
Normal file
71
packages/hooks/hook-protocol/src/detached.ts
Normal file
@@ -0,0 +1,71 @@
|
||||
/**
|
||||
* Quiescence tracking for a bridge's DETACHED hook runs. The waterfall-shaped
|
||||
* hook points (`UserPromptSubmit`, `PreToolUse`, …) are awaited by their seams,
|
||||
* but the emit-shaped points (`SessionStart`, `SubagentStart`, `SubagentStop`)
|
||||
* run fire-and-forget: no seam awaits them, so without tracking a bridge's
|
||||
* disposal could strand a live hook process and let a late continuation fire
|
||||
* into a disposed context (docs/defensive-patterns.md: dispose must reach
|
||||
* quiescence). A bridge creates one tracker in `apply()`, passes
|
||||
* {@link DetachedRuns.signal} to each detached {@link runHook} call, wraps the
|
||||
* full run chain (the hook run PLUS its `.then` continuation) in
|
||||
* {@link DetachedRuns.track}, and registers {@link DetachedRuns.drain} as its
|
||||
* disposer.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-hook-protocol/detached
|
||||
*/
|
||||
|
||||
/** In-flight registry for one bridge's detached hook runs; see the module doc for the wiring contract. */
|
||||
export interface DetachedRuns {
|
||||
/**
|
||||
* The abort signal every tracked run must hand to {@link runHook} (via its
|
||||
* `signal` option). {@link drain} fires it so a still-running hook process is
|
||||
* killed rather than awaited out to its timeout (default 10 minutes).
|
||||
*/
|
||||
readonly signal: AbortSignal
|
||||
/**
|
||||
* Register one detached run until it settles. Pass the FULL chain — the hook
|
||||
* run and its continuation/error handler — so {@link drain} waits for the
|
||||
* side effects (an inject, a warn), not just the process exit. A rejected
|
||||
* chain is absorbed here (settlement bookkeeping only), but rejection
|
||||
* handling is still the caller's job: an untracked `.catch` is what turns a
|
||||
* failure into a logged warning instead of silence.
|
||||
* @param run - the detached run chain to hold until settled.
|
||||
*/
|
||||
track(run: Promise<unknown>): void
|
||||
/**
|
||||
* Abort {@link signal}, then resolve once every tracked chain has settled —
|
||||
* including chains tracked while the drain is in progress. The bridge
|
||||
* registers this as its effect disposer; cordis awaits it, so
|
||||
* `fiber.dispose()` resolving means the bridge's detached work is quiescent.
|
||||
* A run tracked AFTER drain resolves is not awaited by anyone — by then the
|
||||
* bridge's listeners are disposed, so nothing can start one.
|
||||
* @returns resolves when all tracked runs have settled.
|
||||
*/
|
||||
drain(): Promise<void>
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a {@link DetachedRuns} tracker (one per bridge `apply()`); settled
|
||||
* runs are pruned so a long-lived session does not accumulate them.
|
||||
* @returns the tracker.
|
||||
*/
|
||||
export function createDetachedRuns(): DetachedRuns {
|
||||
const inflight = new Set<Promise<unknown>>()
|
||||
const controller = new AbortController()
|
||||
return {
|
||||
signal: controller.signal,
|
||||
track(run: Promise<unknown>): void {
|
||||
inflight.add(run)
|
||||
const settled = (): void => { inflight.delete(run) }
|
||||
void run.then(settled, settled)
|
||||
},
|
||||
async drain(): Promise<void> {
|
||||
controller.abort(new Error('hook bridge disposed'))
|
||||
// Re-check after each wave: a chain can be tracked while a prior wave is
|
||||
// settling; loop until the registry is observed empty.
|
||||
while (inflight.size > 0) {
|
||||
await Promise.allSettled([...inflight])
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -66,6 +66,9 @@ export const DEFAULT_STDERR_SUMMARY_MAX_CHARS = 500
|
||||
* `undefined` when empty, cut at `maxChars` with an ellipsis when over. The
|
||||
* bound is a parameter — like `runHook`'s `defaultTimeoutMs`, each bridge owns
|
||||
* the config default and passes it in.
|
||||
* @param stderr - the hook's raw captured stderr.
|
||||
* @param maxChars - the character cap for the summary (the bridge's config value).
|
||||
* @returns the trimmed, capped summary, or `undefined` when stderr is blank.
|
||||
*/
|
||||
export function summarizeStderr(stderr: string, maxChars: number): string | undefined {
|
||||
const t = stderr.trim()
|
||||
@@ -73,7 +76,11 @@ export function summarizeStderr(stderr: string, maxChars: number): string | unde
|
||||
return t.length > maxChars ? t.slice(0, maxChars) + '…' : t
|
||||
}
|
||||
|
||||
/** Append a `hook/invoked` provenance event to `session`. */
|
||||
/**
|
||||
* Append a `hook/invoked` provenance event to `session`.
|
||||
* @param session - the session whose open turn records the event.
|
||||
* @param invocation - the invocation identity; an absent `matcher` is omitted from the payload.
|
||||
*/
|
||||
export function appendHookInvoked(session: Session, invocation: HookInvocation): void {
|
||||
session.append('hook/invoked', {
|
||||
turn: invocation.turn,
|
||||
@@ -91,6 +98,8 @@ export function appendHookInvoked(session: Session, invocation: HookInvocation):
|
||||
* else `'pass'`; `stderrSummary` is the trimmed stderr truncated to
|
||||
* `record.stderrSummaryMaxChars` characters (omitted when empty); `exitCode`
|
||||
* is omitted when the hook never ran.
|
||||
* @param session - the session whose open turn records the event.
|
||||
* @param record - the outcome to record: the decoded output plus the summary cap and duration.
|
||||
*/
|
||||
export function appendHookResult(session: Session, record: HookResultRecord): void {
|
||||
const { output } = record
|
||||
|
||||
@@ -15,6 +15,8 @@
|
||||
* session-event helpers (declaration-merged into `SessionEventMap`);
|
||||
* `appendHookResult` derives the durable `decision`/`stderrSummary` from the
|
||||
* {@link HookOutput} so the shared event's semantics live in one place.
|
||||
* - {@link createDetachedRuns} — quiescence tracking for the fire-and-forget
|
||||
* hook points: disposal aborts and drains a bridge's detached runs.
|
||||
*
|
||||
* Each bridge owns what genuinely DIFFERS: building the per-event stdin payload
|
||||
* (CC vs Codex field sets), the dialect's env/substitution, and mapping the
|
||||
@@ -38,3 +40,5 @@ export { mergeHookOutputs } from './merge.ts'
|
||||
export type { MergedDecision, MergedHookOutcome } from './merge.ts'
|
||||
export { appendHookInvoked, appendHookResult, DEFAULT_STDERR_SUMMARY_MAX_CHARS, summarizeStderr } from './events.ts'
|
||||
export type { HookInvocation, HookResultRecord } from './events.ts'
|
||||
export { createDetachedRuns } from './detached.ts'
|
||||
export type { DetachedRuns } from './detached.ts'
|
||||
|
||||
@@ -34,6 +34,10 @@ const CLAUDE_LITERAL = /^[A-Za-z0-9_|]+$/
|
||||
* pattern exact-matches the query (splitting `|` into alternatives); every other
|
||||
* `claude` pattern and ALL `codex` patterns are tested as an unanchored regex.
|
||||
* An invalid regex matches nothing (never throws).
|
||||
* @param matcher - the configured pattern; absent/empty/`'*'` are the match-all sentinels.
|
||||
* @param query - the candidate value (a tool name, a session source, …).
|
||||
* @param mode - the dialect deciding literal-vs-regex interpretation of the pattern.
|
||||
* @returns `true` when the pattern selects the query; `false` on a non-match or an invalid regex.
|
||||
*/
|
||||
export function matchesMatcher(matcher: string | undefined, query: string, mode: MatcherMode): boolean {
|
||||
if (isMatchAll(matcher)) return true
|
||||
|
||||
@@ -71,6 +71,8 @@ function decisionForRank(maxRank: number): MergedDecision {
|
||||
* into one {@link MergedHookOutcome} by the precedence rules above. An empty list
|
||||
* yields a neutral outcome (`decision: 'none'`, no stop, empty context) — the
|
||||
* caller treats that as "no hook had anything to say".
|
||||
* @param outputs - every matched hook's decoded output, in hook order.
|
||||
* @returns the single folded outcome the bridge maps onto its seam.
|
||||
*/
|
||||
export function mergeHookOutputs(outputs: HookOutput[]): MergedHookOutcome {
|
||||
let maxRank = 0
|
||||
|
||||
@@ -70,6 +70,11 @@ export interface RunHookResult {
|
||||
* `exitCode: undefined`, so the caller's merge logic treats it as a
|
||||
* non-blocking error rather than crashing the turn. `now` is injected for
|
||||
* testable durations.
|
||||
* @param bash - the executor seam the command runs through.
|
||||
* @param hook - the configured command; its `timeoutSec` (wire unit: seconds) overrides the default timeout.
|
||||
* @param options - the invocation's payload, env, cwd, signal, stdin framing, and default timeout.
|
||||
* @param now - millisecond clock used for the reported duration.
|
||||
* @returns the decoded output plus the run's wall-clock duration.
|
||||
*/
|
||||
export async function runHook(
|
||||
bash: BashExecutor,
|
||||
|
||||
68
packages/hooks/hook-protocol/tests/detached.spec.ts
Normal file
68
packages/hooks/hook-protocol/tests/detached.spec.ts
Normal file
@@ -0,0 +1,68 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { createDetachedRuns } from '@deepseek-ai/dsh-hook-protocol'
|
||||
|
||||
/** A promise settled from outside, so a test controls exactly when a tracked run finishes. */
|
||||
function deferred(): { promise: Promise<void>; resolve: () => void; reject: (error: Error) => void } {
|
||||
let resolve!: () => void
|
||||
let reject!: (error: Error) => void
|
||||
const promise = new Promise<void>((res, rej) => { resolve = res; reject = rej })
|
||||
return { promise, resolve, reject }
|
||||
}
|
||||
|
||||
describe('createDetachedRuns', () => {
|
||||
it('starts with an unfired signal; drain fires it (so still-running hook processes get killed)', async () => {
|
||||
const detached = createDetachedRuns()
|
||||
expect(detached.signal.aborted).toBe(false)
|
||||
await detached.drain()
|
||||
expect(detached.signal.aborted).toBe(true)
|
||||
expect(String(detached.signal.reason)).toContain('hook bridge disposed')
|
||||
})
|
||||
|
||||
it('drain with nothing tracked resolves immediately', async () => {
|
||||
await expect(createDetachedRuns().drain()).resolves.toBeUndefined()
|
||||
})
|
||||
|
||||
it('drain waits for a tracked run to settle', async () => {
|
||||
const detached = createDetachedRuns()
|
||||
const run = deferred()
|
||||
detached.track(run.promise)
|
||||
let drained = false
|
||||
const draining = detached.drain().then(() => { drained = true })
|
||||
// Give the drain every chance to (wrongly) resolve before the run settles.
|
||||
await new Promise(resolve => setTimeout(resolve, 10))
|
||||
expect(drained).toBe(false)
|
||||
run.resolve()
|
||||
await draining
|
||||
expect(drained).toBe(true)
|
||||
})
|
||||
|
||||
it('drain waits for a run tracked WHILE a prior wave was settling', async () => {
|
||||
const detached = createDetachedRuns()
|
||||
const first = deferred()
|
||||
const second = deferred()
|
||||
detached.track(first.promise)
|
||||
// The late run enters the registry from the first run's own continuation —
|
||||
// after drain() snapshotted its first wave.
|
||||
void first.promise.then(() => { detached.track(second.promise) })
|
||||
let drained = false
|
||||
const draining = detached.drain().then(() => { drained = true })
|
||||
first.resolve()
|
||||
await new Promise(resolve => setTimeout(resolve, 10))
|
||||
expect(drained).toBe(false)
|
||||
second.resolve()
|
||||
await draining
|
||||
expect(drained).toBe(true)
|
||||
})
|
||||
|
||||
it('a rejected tracked run is absorbed by the settlement bookkeeping (drain still resolves)', async () => {
|
||||
const detached = createDetachedRuns()
|
||||
const run = deferred()
|
||||
detached.track(run.promise)
|
||||
// The caller-side handler every bridge attaches; the tracker's own
|
||||
// bookkeeping must not depend on it, but an UNHANDLED rejection would fail
|
||||
// the test run, which is exactly the guarantee under test.
|
||||
run.promise.catch(() => {})
|
||||
run.reject(new Error('hook run boom'))
|
||||
await expect(detached.drain()).resolves.toBeUndefined()
|
||||
})
|
||||
})
|
||||
@@ -42,6 +42,8 @@ The hooks **themselves** run in the agent's session workspace: for the agent-sco
|
||||
| `SubagentStart` | `subagent/start` (emit) | additionalContext → `agent.inject()` into the live child |
|
||||
| `SubagentStop` | `subagent/end` (emit) | observe-only |
|
||||
|
||||
The three emit points run detached — no seam awaits a `SessionStart`/`SubagentStart`/`SubagentStop` hook. Each run chain is tracked, and disposing the bridge aborts still-running hook processes, then drains the continuations before the dispose resolves (`createDetachedRuns` in `dsh-hook-protocol`).
|
||||
|
||||
The matcher subject is the tool name (`PreToolUse`/`PostToolUse`), the session source (`SessionStart`), or a constant `agent_type` of `general-purpose` (`SubagentStart`/`SubagentStop` — the harness subagent seam carries no per-kind label, so the bridge reports Claude Code's own Task-tool default; a default/`*`/empty `agent_type` matcher fires, a specific-kind matcher does not); `UserPromptSubmit`/`Stop` ignore matchers. Multiple file-configured hooks on one point run **serially, in config order**, and fold most-restrictively (`deny > ask > allow`, see `dsh-hook-protocol`); serial keeps each hook's `hook/invoked`/`hook/result` pair adjacent in the log, and the fold is order-independent for the decision (see the RFC's "run serially, not concurrently" note).
|
||||
|
||||
## Context source
|
||||
|
||||
@@ -43,7 +43,12 @@ function asObject(value: unknown): Record<string, unknown> | undefined {
|
||||
: undefined
|
||||
}
|
||||
|
||||
/** Apply `${CLAUDE_PLUGIN_ROOT}` / `${CLAUDE_PROJECT_DIR}` substitution to a command string. */
|
||||
/**
|
||||
* Apply `${CLAUDE_PLUGIN_ROOT}` / `${CLAUDE_PROJECT_DIR}` substitution to a command string.
|
||||
* @param command - the raw command from config.
|
||||
* @param vars - the substitution values; a token whose variable is unset stays verbatim.
|
||||
* @returns the command with every occurrence of each set token replaced.
|
||||
*/
|
||||
export function substituteCommand(command: string, vars: SubstitutionVars): string {
|
||||
let out = command
|
||||
if (vars.pluginRoot !== undefined) out = out.split('${CLAUDE_PLUGIN_ROOT}').join(vars.pluginRoot)
|
||||
@@ -57,6 +62,9 @@ export function substituteCommand(command: string, vars: SubstitutionVars): stri
|
||||
* Non-command hooks and malformed entries are dropped (recorded in `skipped` /
|
||||
* silently ignored) rather than throwing — a bad hook config must not crash boot.
|
||||
* `vars` are substituted into every surviving `command`.
|
||||
* @param raw - the parsed JSON config: a settings object with a `hooks` key, or the bare event map.
|
||||
* @param vars - substitution values applied to every surviving `command` (defaults to none).
|
||||
* @returns the runnable per-event groups plus the skipped non-command hooks.
|
||||
*/
|
||||
export function parseClaudeConfig(raw: unknown, vars: SubstitutionVars = {}): ParsedClaudeConfig {
|
||||
const config: ClaudeHookConfig = {}
|
||||
|
||||
@@ -31,6 +31,7 @@ import type { PostToolDecision, PreToolDecision, ToolExecution, ToolExecutionRes
|
||||
import {
|
||||
appendHookInvoked,
|
||||
appendHookResult,
|
||||
createDetachedRuns,
|
||||
DEFAULT_HOOK_TIMEOUT_MS,
|
||||
DEFAULT_STDERR_SUMMARY_MAX_CHARS,
|
||||
matchesMatcher,
|
||||
@@ -128,6 +129,14 @@ export function apply(ctx: Context, config: Config): void {
|
||||
return
|
||||
}
|
||||
|
||||
// --- The emit-shaped points (SessionStart, SubagentStart, SubagentStop) run
|
||||
// detached — no seam awaits them — so every run chain is tracked and disposal
|
||||
// aborts still-running hook processes, then drains the continuations
|
||||
// (docs/defensive-patterns.md: dispose must reach quiescence). After the parse
|
||||
// gate: a bridge that registered nothing has nothing to drain. ---
|
||||
const detached = createDetachedRuns()
|
||||
ctx.effect(() => () => detached.drain(), 'hooks-claude: drain detached hook runs')
|
||||
|
||||
/**
|
||||
* Run every command hook configured for `point` whose matcher selects
|
||||
* `matchQuery`, with the per-event `payload` on stdin, and fold the results.
|
||||
@@ -237,14 +246,14 @@ export function apply(ctx: Context, config: Config): void {
|
||||
// to the interception seams; today the contract is "injected as soon as the
|
||||
// hook resolves", not "before the first request". ---
|
||||
ctx.on('agent/session-start', (agent, source) => {
|
||||
void runPoint('SessionStart', source, sessionStartPayload(agent, source), { agent })
|
||||
detached.track(runPoint('SessionStart', source, sessionStartPayload(agent, source), { agent, signal: detached.signal })
|
||||
.then((merged) => {
|
||||
const context = contextFrom(merged)
|
||||
if (context) agent.inject(context.content, { source: context.source })
|
||||
})
|
||||
.catch((error: unknown) => {
|
||||
ctx.logger.warn(`hooks-claude: SessionStart hook failed: ${String(error)}`)
|
||||
})
|
||||
}))
|
||||
})
|
||||
|
||||
// --- UserPromptSubmit → PromptDecision. The prompt text is the payload; no
|
||||
@@ -330,12 +339,12 @@ export function apply(ctx: Context, config: Config): void {
|
||||
// a specific-kind matcher does not (documented in the RFC). ---
|
||||
ctx.on('subagent/start', (info) => {
|
||||
const child = ctx.get('agents')?.get(info.id)
|
||||
void runPoint('SubagentStart', SUBAGENT_TYPE, subagentPayload('SubagentStart', info, child), { ...child ? { agent: child } : {} })
|
||||
detached.track(runPoint('SubagentStart', SUBAGENT_TYPE, subagentPayload('SubagentStart', info, child), { ...child ? { agent: child } : {}, signal: detached.signal })
|
||||
.then((merged) => {
|
||||
const context = contextFrom(merged)
|
||||
if (context && child) child.inject(context.content, { source: context.source })
|
||||
})
|
||||
.catch((error: unknown) => { ctx.logger.warn(`hooks-claude: SubagentStart hook failed: ${String(error)}`) })
|
||||
.catch((error: unknown) => { ctx.logger.warn(`hooks-claude: SubagentStart hook failed: ${String(error)}`) }))
|
||||
})
|
||||
ctx.on('subagent/end', (info) => {
|
||||
// Look up the child (still recoverable: `subagent/end` fires from the
|
||||
@@ -343,9 +352,10 @@ export function apply(ctx: Context, config: Config): void {
|
||||
// disposes it) so the hook runs in the child's cwd, not the server default.
|
||||
// No `.then`/inject follows (SubagentStop only observes), and no `turn` is
|
||||
// passed (so no `hook/*` log records), so runPoint has nothing that can
|
||||
// reject — no `.catch` is needed. Fire-and-forget.
|
||||
// reject — no `.catch` is needed (the tracker's settlement bookkeeping
|
||||
// would absorb one anyway).
|
||||
const child = ctx.get('agents')?.get(info.id)
|
||||
void runPoint('SubagentStop', SUBAGENT_TYPE, subagentPayload('SubagentStop', info, child), { ...child ? { agent: child } : {} })
|
||||
detached.track(runPoint('SubagentStop', SUBAGENT_TYPE, subagentPayload('SubagentStop', info, child), { ...child ? { agent: child } : {}, signal: detached.signal }))
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { mkdtempSync, rmSync, writeFileSync, chmodSync } from 'node:fs'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { chmodSync, existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { Context } from 'cordis'
|
||||
import { Context, type Fiber } from 'cordis'
|
||||
import Loader from '@cordisjs/plugin-loader'
|
||||
import LlmService from '@deepseek-ai/dsh-llm'
|
||||
import SessionStore, { type SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
@@ -39,6 +39,11 @@ function writeConfig(hooks: unknown, scripts: Record<string, string> = {}): stri
|
||||
}
|
||||
|
||||
async function harness(configDir: string, adapter: MockAdapter): Promise<Context> {
|
||||
return (await harnessWithFiber(configDir, adapter)).ctx
|
||||
}
|
||||
|
||||
/** {@link harness}, also exposing the bridge's fiber for tests that dispose it. */
|
||||
async function harnessWithFiber(configDir: string, adapter: MockAdapter): Promise<{ ctx: Context; hooks: Fiber }> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(SessionStore)
|
||||
@@ -47,9 +52,9 @@ async function harness(configDir: string, adapter: MockAdapter): Promise<Context
|
||||
await ctx.plugin(AgentRegistry)
|
||||
await ctx.plugin(AgentLoop, { agents: [] })
|
||||
await ctx.plugin(LocalBashExecutor, { timeoutMs: 10_000 })
|
||||
await ctx.plugin(HooksClaude, { configPath: join(configDir, 'hooks.json') })
|
||||
const hooks = await ctx.plugin(HooksClaude, { configPath: join(configDir, 'hooks.json') })
|
||||
ctx.llm.registerAdapter(['mock'], adapter)
|
||||
return ctx
|
||||
return { ctx, hooks }
|
||||
}
|
||||
|
||||
function waitForIdle(ctx: Context, agent: ReactLoopAgent): Promise<void> {
|
||||
@@ -285,19 +290,59 @@ describe('hooks-claude bridge — SubagentStart / SubagentStop (observe)', () =>
|
||||
} }))
|
||||
|
||||
const adapter = new MockAdapter([])
|
||||
const ctx = await harness(dir, adapter)
|
||||
const { ctx, hooks } = await harnessWithFiber(dir, adapter)
|
||||
// Drive the observe-only lifecycle events directly (no real child needed — the
|
||||
// bridge just listens). The agents registry is absent here, so SubagentStart's
|
||||
// bridge just listens). No child agent is registered, so SubagentStart's
|
||||
// child lookup yields undefined and it simply runs the hook.
|
||||
ctx.emit('subagent/start', { provider: 'inproc', id: AgentId('child-1') })
|
||||
ctx.emit('subagent/end', { provider: 'inproc', id: AgentId('child-1'), stopReason: 'completed', lastAssistantMessage: [{ type: 'text', text: 'done' }] })
|
||||
|
||||
// Both hooks run async (detached .then); poll for their marker files rather
|
||||
// than a fixed sleep that flakes under load.
|
||||
const { existsSync } = await import('node:fs')
|
||||
await waitFor(() => existsSync(startMarker) && existsSync(stopMarker))
|
||||
expect(existsSync(startMarker)).toBe(true)
|
||||
expect(existsSync(stopMarker)).toBe(true)
|
||||
// The markers prove the hook PROCESSES ran, not that the detached `.then`
|
||||
// continuations did (`touch` lands before the process exits). Dispose drains
|
||||
// them, so the no-context arm of the SubagentStart continuation — covered
|
||||
// only here — executes before this file's coverage snapshot instead of
|
||||
// racing it (the arm went uncovered on a loaded CI runner and failed the
|
||||
// per-file 100% branch gate).
|
||||
await hooks.dispose()
|
||||
})
|
||||
|
||||
it('disposing the bridge aborts a still-running hook and drains to quiescence', async () => {
|
||||
const dir = mkdtempSync(join(tmpdir(), 'dsh-hooks-claude-'))
|
||||
dirs.push(dir)
|
||||
const pidFile = join(dir, 'pid')
|
||||
const marker = join(dir, 'started')
|
||||
const slowHook = join(dir, 'slow.sh')
|
||||
// Record the hook shell's PID and touch the marker FIRST so the test can
|
||||
// tell "the hook is genuinely mid-run", then sleep far past the suite
|
||||
// timeout. Dispose must KILL the process (the tracker's abort signal), not
|
||||
// await its exit or its 10-minute default hook timeout.
|
||||
writeFileSync(slowHook, `#!/usr/bin/env bash\necho $$ > "${pidFile}"\ntouch "${marker}"\nsleep 30\n`)
|
||||
chmodSync(slowHook, 0o755)
|
||||
writeFileSync(join(dir, 'hooks.json'), JSON.stringify({ hooks: {
|
||||
SubagentStart: [{ hooks: [{ type: 'command', command: slowHook }] }],
|
||||
} }))
|
||||
|
||||
const { ctx, hooks } = await harnessWithFiber(dir, new MockAdapter([]))
|
||||
const warn = vi.fn()
|
||||
ctx.logger.warn = warn as never
|
||||
ctx.emit('subagent/start', { provider: 'inproc', id: AgentId('child-1') })
|
||||
await waitFor(() => existsSync(marker))
|
||||
const pid = Number(readFileSync(pidFile, 'utf8').trim())
|
||||
await hooks.dispose()
|
||||
// Quiescence, not just promptness: the drain resolves only after the run
|
||||
// settled, and the run settles only after the killed process was reaped —
|
||||
// so by the time dispose returns, the PID must be GONE (kill(pid, 0)
|
||||
// throws ESRCH). An untracked fire-and-forget regression would leave the
|
||||
// process alive (or unreaped) and fail this deterministically.
|
||||
expect(() => process.kill(pid, 0)).toThrow()
|
||||
// The aborted run resolves as a non-blocking error (runHook never rejects),
|
||||
// so the drained continuation must NOT have logged a failure.
|
||||
expect(warn).not.toHaveBeenCalledWith(expect.stringContaining('SubagentStart hook failed'))
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -48,6 +48,8 @@ The hooks themselves run in the agent's session workspace: for the agent-scoped
|
||||
|
||||
A tool call's payload carries the real `tool_name` (the same value the matcher tests) and Codex's `tool_input: { command }` shape (the `command` arg when present, else `''`). The matcher subject is the tool name (`PreToolUse`/`PostToolUse`) or the session source (`SessionStart`); `UserPromptSubmit`/`Stop` ignore matchers.
|
||||
|
||||
`SessionStart` — the one emit point — runs detached; each run chain is tracked, and disposing the bridge aborts a still-running hook process, then drains the continuation before the dispose resolves (`createDetachedRuns` in `dsh-hook-protocol`).
|
||||
|
||||
## Context source
|
||||
|
||||
Injected context carries an explicit `{ kind: 'plugin', plugin: 'hooks-codex' }` source (`agent.inject()` would otherwise default it to `{ kind: 'user' }`).
|
||||
|
||||
@@ -41,6 +41,8 @@ function asObject(value: unknown): Record<string, unknown> | undefined {
|
||||
* `type !== 'command'` and `async: true` command hooks are skipped (recorded in
|
||||
* `skipped`). Malformed entries are ignored rather than thrown — a bad config
|
||||
* must not crash boot. No command substitution (Codex does none).
|
||||
* @param raw - the parsed JSON config: a `{ hooks: … }` wrapper or the bare event map.
|
||||
* @returns the runnable per-event groups plus the skipped hooks with their reasons.
|
||||
*/
|
||||
export function parseCodexConfig(raw: unknown): ParsedCodexConfig {
|
||||
const config: CodexHookConfig = {}
|
||||
|
||||
@@ -24,6 +24,7 @@ import type { PostToolDecision, PreToolDecision, ToolExecution, ToolExecutionRes
|
||||
import {
|
||||
appendHookInvoked,
|
||||
appendHookResult,
|
||||
createDetachedRuns,
|
||||
DEFAULT_HOOK_TIMEOUT_MS,
|
||||
DEFAULT_STDERR_SUMMARY_MAX_CHARS,
|
||||
matchesMatcher,
|
||||
@@ -97,6 +98,12 @@ export function apply(ctx: Context, config: Config): void {
|
||||
|
||||
const model = config.model ?? ''
|
||||
|
||||
// SessionStart is the one emit-shaped (detached) point Codex has: track its
|
||||
// run chains so disposal aborts a still-running hook process and drains the
|
||||
// continuation (docs/defensive-patterns.md: dispose must reach quiescence).
|
||||
const detached = createDetachedRuns()
|
||||
ctx.effect(() => () => detached.drain(), 'hooks-codex: drain detached hook runs')
|
||||
|
||||
async function runPoint(
|
||||
point: string,
|
||||
matchQuery: string,
|
||||
@@ -189,12 +196,12 @@ export function apply(ctx: Context, config: Config): void {
|
||||
// the model (a slow hook can miss the first request). Gating is a deferred
|
||||
// loop-level change; the contract is "injected as soon as the hook resolves".
|
||||
ctx.on('agent/session-start', (agent, source) => {
|
||||
void runPoint('SessionStart', source, { ...base(agent, 'SessionStart', model), source }, { agent, plainStdoutAsContext: true })
|
||||
detached.track(runPoint('SessionStart', source, { ...base(agent, 'SessionStart', model), source }, { agent, plainStdoutAsContext: true, signal: detached.signal })
|
||||
.then((merged) => {
|
||||
const context = contextFrom(merged)
|
||||
if (context) agent.inject(context.content, { source: context.source })
|
||||
})
|
||||
.catch((error: unknown) => { ctx.logger.warn(`hooks-codex: SessionStart hook failed: ${String(error)}`) })
|
||||
.catch((error: unknown) => { ctx.logger.warn(`hooks-codex: SessionStart hook failed: ${String(error)}`) }))
|
||||
})
|
||||
|
||||
// UserPromptSubmit → PromptDecision. Codex can only BLOCK (no allow/ask).
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { mkdtempSync, rmSync, writeFileSync, chmodSync } from 'node:fs'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { chmodSync, existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { Context } from 'cordis'
|
||||
@@ -62,6 +62,15 @@ function waitForIdle(ctx: Context, agent: ReactLoopAgent): Promise<void> {
|
||||
}
|
||||
function events(agent: ReactLoopAgent): SessionEvent[] { return [...agent.session.events] }
|
||||
|
||||
/** Poll `predicate` until true or the deadline passes (detached hook effects can't be awaited directly). */
|
||||
async function waitFor(predicate: () => boolean, timeout = 5000, interval = 10): Promise<void> {
|
||||
const deadline = Date.now() + timeout
|
||||
while (!predicate()) {
|
||||
if (Date.now() > deadline) throw new Error('waitFor: condition not met before deadline')
|
||||
await new Promise(r => setTimeout(r, interval))
|
||||
}
|
||||
}
|
||||
|
||||
describe('hooks-codex bridge', () => {
|
||||
it('a PreToolUse hook (exit 2) denies a tool the regex matcher matches as a substring', async () => {
|
||||
const dir = configDir()
|
||||
@@ -159,6 +168,43 @@ describe('hooks-codex bridge', () => {
|
||||
expect(events(agent).some(e => e.type === 'hook/invoked')).toBe(false) // no hook ran
|
||||
})
|
||||
|
||||
it('disposing the bridge aborts a still-running SessionStart hook and drains to quiescence', async () => {
|
||||
const dir = configDir()
|
||||
const pidFile = join(dir, 'pid')
|
||||
const marker = join(dir, 'started')
|
||||
// Record the hook shell's PID and touch the marker FIRST so the test can
|
||||
// tell "the hook is genuinely mid-run", then sleep far past the suite
|
||||
// timeout. Dispose must KILL the process (the tracker's abort signal wired
|
||||
// through this bridge's runPoint), not await its exit.
|
||||
const slow = script(dir, 'slow.sh', `#!/usr/bin/env bash\necho $$ > "${pidFile}"\ntouch "${marker}"\nsleep 30\n`)
|
||||
writeHooks(dir, { SessionStart: [{ hooks: [{ type: 'command', command: slow }] }] })
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await ctx.plugin(AgentRegistry)
|
||||
await ctx.plugin(AgentLoop, { agents: [] })
|
||||
await ctx.plugin(LocalBashExecutor, { timeoutMs: 10_000 })
|
||||
const fiber = await ctx.plugin(HooksCodex, { configPath: join(dir, 'hooks.json'), model: 'm' })
|
||||
ctx.llm.registerAdapter(['mock'], new MockAdapter([]))
|
||||
const warn = vi.fn()
|
||||
ctx.logger.warn = warn as never
|
||||
ctx.agentLoop.create(AgentId('a1'), { model: 'mock' }) // fires agent/session-start
|
||||
await waitFor(() => existsSync(marker))
|
||||
const pid = Number(readFileSync(pidFile, 'utf8').trim())
|
||||
await fiber.dispose()
|
||||
// Quiescence, not just promptness: the drain resolves only after the run
|
||||
// settled, and the run settles only after the killed process was reaped —
|
||||
// so by the time dispose returns, the PID must be GONE (kill(pid, 0)
|
||||
// throws ESRCH). An untracked fire-and-forget regression would leave the
|
||||
// process alive (or unreaped) and fail this deterministically.
|
||||
expect(() => process.kill(pid, 0)).toThrow()
|
||||
// The aborted run resolves as a non-blocking error (runHook never rejects),
|
||||
// so the drained continuation must NOT have logged a failure.
|
||||
expect(warn).not.toHaveBeenCalledWith(expect.stringContaining('SessionStart hook failed'))
|
||||
})
|
||||
|
||||
it('has the namespace-plugin export shape (no stray default) so the Loader keeps name/inject/apply', () => {
|
||||
expect('default' in HooksCodex).toBe(false)
|
||||
expect(HooksCodex.name).toBe('hooks-codex')
|
||||
|
||||
@@ -13,7 +13,9 @@ import { parseSse } from './sse.ts'
|
||||
import { translate } from './translate.ts'
|
||||
import type { WireError } from './types.ts'
|
||||
|
||||
/** Constructor options for {@link DeepSeekAdapter}; the plugin's `apply` resolves them from Config + environment. */
|
||||
export interface DeepSeekAdapterOptions {
|
||||
/** Bearer token sent in the `authorization` header on every request. */
|
||||
apiKey: string
|
||||
/** Endpoint base; `/chat/completions` is appended. */
|
||||
baseURL: string
|
||||
@@ -21,7 +23,11 @@ export interface DeepSeekAdapterOptions {
|
||||
defaults?: RequestDefaults
|
||||
}
|
||||
|
||||
/** Map an HTTP status to a stable LlmError code. */
|
||||
/**
|
||||
* Map an HTTP status to a stable LlmError code.
|
||||
* @param status - status of a non-2xx provider response.
|
||||
* @returns `AUTH` (401/403), `RATE_LIMIT` (429), `INVALID_REQUEST` (400), `SERVER` (5xx), or `HTTP_<status>` for anything else.
|
||||
*/
|
||||
export function httpErrorCode(status: number): string {
|
||||
if (status === 401 || status === 403) return 'AUTH'
|
||||
if (status === 429) return 'RATE_LIMIT'
|
||||
|
||||
@@ -34,6 +34,12 @@ export type * from './types.ts'
|
||||
export const name = 'llm-deepseek'
|
||||
export const inject = ['llm']
|
||||
|
||||
/**
|
||||
* Plugin config, validated by the same-named schemastery schema. Every field
|
||||
* is optional in yml: credentials/endpoint fall back to the environment (a
|
||||
* missing API key fails plugin load, not the first call), and omitted
|
||||
* thinking fields send nothing on the wire, so the provider default applies.
|
||||
*/
|
||||
export interface Config {
|
||||
/** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */
|
||||
apiKey?: string
|
||||
|
||||
@@ -66,6 +66,8 @@ function serializeAssistant(message: Message): WireMessage {
|
||||
* `{role: 'tool'}` messages; the harness puts each tool result in its own
|
||||
* user-role message, so a mixed user message contributes its text first and
|
||||
* its tool results as separate wire messages after.
|
||||
* @param messages - the harness conversation, in order.
|
||||
* @returns the wire messages; order preserved, each tool result expanded into its own entry.
|
||||
*/
|
||||
export function serializeMessages(messages: Message[]): WireMessage[] {
|
||||
const wire: WireMessage[] = []
|
||||
@@ -97,7 +99,14 @@ export function serializeMessages(messages: Message[]): WireMessage[] {
|
||||
return wire
|
||||
}
|
||||
|
||||
/** Build the full wire request. */
|
||||
/**
|
||||
* Build the full wire request. Always streaming (`stream: true`, usage
|
||||
* reporting on); optional fields are omitted rather than sent as null, so
|
||||
* provider defaults apply.
|
||||
* @param options - the harness request (model, history, system, tools, sampling).
|
||||
* @param defaults - adapter-level thinking defaults; undefined fields put nothing on the wire.
|
||||
* @returns the chat-completions request body.
|
||||
*/
|
||||
export function serializeRequest(options: GenerateOptions, defaults: RequestDefaults = {}): WireRequest {
|
||||
const messages: WireMessage[] = []
|
||||
if (options.system !== undefined) {
|
||||
|
||||
@@ -37,6 +37,8 @@ function eventData(block: string): string | undefined {
|
||||
* Parse a byte stream into SSE data payloads. Yields `[DONE]` as the final
|
||||
* value and returns; throws `LlmError('STREAM_CLOSED')` when the stream ends
|
||||
* without it (truncated response — the model call cannot be trusted).
|
||||
* @param stream - raw SSE bytes; reads may split anywhere, including mid-UTF-8 sequence.
|
||||
* @returns each event's data payload in arrival order, the `[DONE]` sentinel last.
|
||||
*/
|
||||
export async function* parseSse(stream: AsyncIterable<Uint8Array>): AsyncGenerator<string> {
|
||||
const decoder = new TextDecoder()
|
||||
|
||||
@@ -29,7 +29,11 @@ interface OpenBlock {
|
||||
name?: string
|
||||
}
|
||||
|
||||
/** Map the wire finish_reason vocabulary to the harness FinishReason. */
|
||||
/**
|
||||
* Map the wire finish_reason vocabulary to the harness FinishReason.
|
||||
* @param reason - the wire `finish_reason` string.
|
||||
* @returns the mapped reason; unrecognized values (content_filter, …) become `{kind: 'error'}` with the uppercased value as `code`.
|
||||
*/
|
||||
export function mapFinishReason(reason: string): FinishReason {
|
||||
switch (reason) {
|
||||
case 'stop': return { kind: 'stop' }
|
||||
@@ -46,6 +50,8 @@ export function mapFinishReason(reason: string): FinishReason {
|
||||
* (`prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens`,
|
||||
* api/create-chat-completion); the harness TokenUsage convention is
|
||||
* DISJOINT counts, so cache reads are subtracted out of `inputTokens`.
|
||||
* @param usage - wire usage from the finish chunk or the trailing usage-only chunk.
|
||||
* @returns disjoint harness counts; cache/reasoning fields present only when the wire reported them.
|
||||
*/
|
||||
export function mapUsage(usage: WireUsage): TokenUsage {
|
||||
const cacheRead = usage.prompt_tokens_details?.cached_tokens ?? usage.prompt_cache_hit_tokens
|
||||
@@ -75,6 +81,8 @@ function closeBlock(block: OpenBlock): ContentBlock {
|
||||
/**
|
||||
* Consume SSE data payloads (ending with `[DONE]`) and yield StreamChunks.
|
||||
* Malformed JSON payloads abort the stream with `MALFORMED_RESPONSE`.
|
||||
* @param payloads - SSE data payloads from {@link parseSse}, `[DONE]`-terminated.
|
||||
* @returns deltas as they arrive; `block-end`s, `usage`, and `finish` are all deferred to the `[DONE]` sentinel.
|
||||
*/
|
||||
export async function* translate(payloads: AsyncIterable<string>): AsyncGenerator<StreamChunk> {
|
||||
let nextIndex = 0
|
||||
|
||||
@@ -48,12 +48,18 @@ export interface WireToolMessage {
|
||||
content: string
|
||||
}
|
||||
|
||||
/** One entry of the request `messages` array, discriminated on `role`. */
|
||||
export type WireMessage =
|
||||
| WireSystemMessage
|
||||
| WireUserMessage
|
||||
| WireAssistantMessage
|
||||
| WireToolMessage
|
||||
|
||||
/**
|
||||
* Assistant-role history message. The harness replays `content: ""` (never
|
||||
* null) on tool-call-only turns — some gateways reject null — and sends null
|
||||
* only when the turn carried neither text nor tool calls.
|
||||
*/
|
||||
export interface WireAssistantMessage {
|
||||
role: 'assistant'
|
||||
content: string | null
|
||||
@@ -66,12 +72,14 @@ export interface WireAssistantMessage {
|
||||
tool_calls?: WireToolCall[]
|
||||
}
|
||||
|
||||
/** A completed tool call replayed on an assistant history message; `arguments` is the raw JSON string. */
|
||||
export interface WireToolCall {
|
||||
id: string
|
||||
type: 'function'
|
||||
function: { name: string; arguments: string }
|
||||
}
|
||||
|
||||
/** One entry of the request `tools` array; `parameters` is a JSON Schema object. */
|
||||
export interface WireTool {
|
||||
type: 'function'
|
||||
function: {
|
||||
@@ -88,11 +96,13 @@ export interface WireChunk {
|
||||
usage?: WireUsage | null
|
||||
}
|
||||
|
||||
/** One streamed choice (requests always ask for a single one); `finish_reason` is non-null only on its terminal chunk. */
|
||||
export interface WireChoice {
|
||||
delta?: WireDelta
|
||||
finish_reason?: string | null
|
||||
}
|
||||
|
||||
/** The incremental content of one streamed choice; any subset of fields may be present per chunk. */
|
||||
export interface WireDelta {
|
||||
role?: string
|
||||
/** Visible text. Null/empty on reasoning/tool-call chunks. */
|
||||
@@ -105,6 +115,7 @@ export interface WireDelta {
|
||||
tool_calls?: WireToolCallDelta[]
|
||||
}
|
||||
|
||||
/** A streamed fragment of one tool call; fragments sharing an `index` concatenate into one call. */
|
||||
export interface WireToolCallDelta {
|
||||
/** Disambiguates parallel tool calls; stable across a call's deltas. */
|
||||
index: number
|
||||
@@ -119,6 +130,13 @@ export interface WireToolCallDelta {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wire token accounting. `prompt_tokens` INCLUDES cache hits (it equals
|
||||
* `prompt_cache_hit_tokens + prompt_cache_miss_tokens`); `mapUsage` subtracts
|
||||
* them to keep the harness convention of disjoint counts.
|
||||
* `prompt_tokens_details.cached_tokens` is the OpenAI-compat spelling of the
|
||||
* hit count.
|
||||
*/
|
||||
export interface WireUsage {
|
||||
prompt_tokens: number
|
||||
completion_tokens: number
|
||||
|
||||
@@ -21,14 +21,22 @@ import { toPiContext, toStreamChunks } from './convert.ts'
|
||||
/** Reasoning levels surfaced by this adapter (DeepSeek wire: high|max). */
|
||||
export type PiAiReasoning = 'off' | 'high' | 'xhigh'
|
||||
|
||||
/** Constructor options for {@link PiAiAdapter}; the plugin's `apply` resolves them from Config + environment. */
|
||||
export interface PiAiAdapterOptions {
|
||||
/** Bearer token pi-ai sends on every request. */
|
||||
apiKey: string
|
||||
/** Endpoint base; `/chat/completions` is appended. */
|
||||
baseURL: string
|
||||
/** Thinking level applied to every request ('off' disables thinking). */
|
||||
reasoning?: PiAiReasoning | undefined
|
||||
}
|
||||
|
||||
/** Build the inline pi-ai model descriptor for one DeepSeek model name. */
|
||||
/**
|
||||
* Build the inline pi-ai model descriptor for one DeepSeek model name.
|
||||
* @param modelId - harness model name; sent verbatim on the wire.
|
||||
* @param options - adapter options; only `baseURL` is read here (key and reasoning apply per request, not per descriptor).
|
||||
* @returns a descriptor with every DeepSeek compat flag explicit — pi-ai's URL-based auto-detection is never relied on.
|
||||
*/
|
||||
export function buildModel(modelId: string, options: PiAiAdapterOptions): Model<'openai-completions'> {
|
||||
return {
|
||||
id: modelId,
|
||||
|
||||
@@ -55,6 +55,8 @@ function parseArguments(raw: string): Record<string, unknown> {
|
||||
* NAME (pi-ai's `toolName`), which the harness doesn't carry on the result
|
||||
* block — it is recovered from the preceding assistant tool-call with the
|
||||
* same id.
|
||||
* @param options - the harness request; `options.system` maps to pi-ai's single `systemPrompt` slot.
|
||||
* @returns the pi-ai context; `tools` is omitted entirely when the request declares none.
|
||||
*/
|
||||
export function toPiContext(options: GenerateOptions): PiContext {
|
||||
const toolNames = new Map<CallId, string>()
|
||||
@@ -159,7 +161,11 @@ function emptyPiUsage(): PiUsage {
|
||||
}
|
||||
}
|
||||
|
||||
/** Map pi-ai usage (reasoning folded into output by pi-ai). */
|
||||
/**
|
||||
* Map pi-ai usage (reasoning folded into output by pi-ai).
|
||||
* @param usage - cumulative usage from the terminal pi-ai event.
|
||||
* @returns harness counts; cache fields appear only when non-zero (pi-ai reports zeros, not absence).
|
||||
*/
|
||||
export function mapUsage(usage: PiUsage): TokenUsage {
|
||||
return {
|
||||
inputTokens: usage.input,
|
||||
@@ -177,7 +183,11 @@ function classifyPiAiError(message: string): string {
|
||||
return 'PI_AI_ERROR'
|
||||
}
|
||||
|
||||
/** Map a terminal pi-ai event to the harness finish reason. */
|
||||
/**
|
||||
* Map a terminal pi-ai event to the harness finish reason.
|
||||
* @param message - the assistant message carried by the `done` or `error` event.
|
||||
* @returns the harness reason; `error` yields `{kind: 'error'}` with a code classified from the error text.
|
||||
*/
|
||||
export function mapStopReason(message: AssistantMessage): FinishReason {
|
||||
switch (message.stopReason) {
|
||||
case 'stop': return { kind: 'stop' }
|
||||
@@ -195,6 +205,9 @@ export function mapStopReason(message: AssistantMessage): FinishReason {
|
||||
* Translate the pi-ai event stream into StreamChunks. pi-ai never throws
|
||||
* mid-stream — failures arrive as `error` events, which become error/aborted
|
||||
* `finish` chunks (the harness protocol's other error-delivery style).
|
||||
* @param events - one assistant turn's pi-ai event stream.
|
||||
* @returns the harness chunks, ending with `usage` then `finish`; throws
|
||||
* `LlmError` (`STREAM_CLOSED`) if the source ends without a terminal event.
|
||||
*/
|
||||
export async function* toStreamChunks(events: AsyncIterable<AssistantMessageEvent>): AsyncGenerator<StreamChunk> {
|
||||
// pi-ai contentIndex ↔ our block index map 1:1 (both count blocks from 0
|
||||
|
||||
@@ -29,6 +29,11 @@ export { mapStopReason, mapUsage, toPiContext, toStreamChunks } from './convert.
|
||||
export const name = 'llm-pi-ai'
|
||||
export const inject = ['llm']
|
||||
|
||||
/**
|
||||
* Plugin config, validated by the same-named schemastery schema. Every field
|
||||
* is optional in yml: credentials/endpoint fall back to the environment (a
|
||||
* missing API key fails plugin load, not the first call).
|
||||
*/
|
||||
export interface Config {
|
||||
/** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */
|
||||
apiKey?: string
|
||||
|
||||
@@ -40,6 +40,8 @@ export class BlockAssembler {
|
||||
/**
|
||||
* Feed one chunk. Returns the completed block when the chunk closes one
|
||||
* (an explicit `block-end`), otherwise undefined.
|
||||
* @param chunk - the next raw chunk, in stream order.
|
||||
* @returns the authoritative block from the first `block-end` at its index; undefined for every other chunk.
|
||||
*/
|
||||
push(chunk: StreamChunk): ContentBlock | undefined {
|
||||
switch (chunk.type) {
|
||||
@@ -123,20 +125,29 @@ export class BlockAssembler {
|
||||
return partial
|
||||
}
|
||||
|
||||
/** Assemble all blocks seen so far, in stream order. */
|
||||
/**
|
||||
* Assemble all blocks seen so far, in stream order.
|
||||
* @returns one block per seen index; an open block assembles from its
|
||||
* accumulated deltas (an unknown block type never closed by `block-end` throws).
|
||||
*/
|
||||
blocks(): ContentBlock[] {
|
||||
return this.order.map(index => this.assemble(this.mustGet(index), index))
|
||||
}
|
||||
|
||||
/** Usage from the `usage` chunk; undefined until one arrives. */
|
||||
get usage(): TokenUsage | undefined {
|
||||
return this._usage
|
||||
}
|
||||
|
||||
/** Finish reason from the `finish` chunk; `{kind: 'stop'}` when the stream ended without one. */
|
||||
get finish(): FinishReason {
|
||||
return this._finish ?? { kind: 'stop' }
|
||||
}
|
||||
|
||||
/** The assembled assistant message. */
|
||||
/**
|
||||
* The assembled assistant message.
|
||||
* @returns an assistant-role message over `blocks()` (same open-block assembly rules).
|
||||
*/
|
||||
message(): Message {
|
||||
return { role: 'assistant', content: this.blocks() }
|
||||
}
|
||||
|
||||
@@ -54,6 +54,8 @@ export const APP_IDENTITY: AppIdentity = {
|
||||
* The standard `User-Agent` value: `product/version (+url)`. The
|
||||
* parenthesized `+url` comment is the conventional self-identification form
|
||||
* (RFC 9110 §10.1.5 product + comment syntax).
|
||||
* @param identity - the identity to render; defaults to {@link APP_IDENTITY}.
|
||||
* @returns the ready-to-send header value.
|
||||
*/
|
||||
export function userAgent(identity: AppIdentity = APP_IDENTITY): string {
|
||||
return `${identity.product}/${identity.version} (+${identity.url})`
|
||||
@@ -63,6 +65,8 @@ export function userAgent(identity: AppIdentity = APP_IDENTITY): string {
|
||||
* Build the attribution headers an adapter must send on every provider
|
||||
* request. Header names are lowercase (HTTP field names are case-insensitive
|
||||
* on the wire).
|
||||
* @param identity - the identity to send; defaults to {@link APP_IDENTITY} — omission cannot suppress attribution.
|
||||
* @returns headers to merge into the provider request (currently just `user-agent`).
|
||||
*/
|
||||
export function attributionHeaders(
|
||||
identity: AppIdentity = APP_IDENTITY,
|
||||
|
||||
@@ -17,7 +17,11 @@ import type { Branded } from '@deepseek-ai/dsh-brand'
|
||||
*/
|
||||
export type CallId = Branded<'CallId'>
|
||||
|
||||
/** Brand a string as a {@link CallId}. */
|
||||
/**
|
||||
* Brand a string as a {@link CallId}.
|
||||
* @param id - the provider-issued (or synthesized) call id.
|
||||
* @returns the same string, branded; no validation is performed.
|
||||
*/
|
||||
export function CallId(id: string): CallId {
|
||||
return id as CallId
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@
|
||||
* `ErrorOptions`. `name` defaults to the subclass constructor name.
|
||||
*/
|
||||
export class HarnessError extends Error {
|
||||
/** Stable machine-routable failure class (e.g. `RATE_LIMIT`); route on this, never by parsing `message`. */
|
||||
readonly code: string
|
||||
|
||||
constructor(message: string, code: string, options?: ErrorOptions) {
|
||||
@@ -27,7 +28,11 @@ export class HarnessError extends Error {
|
||||
}
|
||||
}
|
||||
|
||||
/** Narrow an arbitrary thrown value to a HarnessError (for `instanceof` at seams). */
|
||||
/**
|
||||
* Narrow an arbitrary thrown value to a HarnessError (for `instanceof` at seams).
|
||||
* @param value - the caught value (`unknown` in catch clauses).
|
||||
* @returns true only for real instances; duck-typed or cross-realm errors do not narrow.
|
||||
*/
|
||||
export function isHarnessError(value: unknown): value is HarnessError {
|
||||
return value instanceof HarnessError
|
||||
}
|
||||
|
||||
@@ -73,7 +73,11 @@ export class LlmError extends HarnessError {
|
||||
* same value to the wire.
|
||||
*/
|
||||
export abstract class LlmAdapter {
|
||||
/** Stream one model call as raw chunks. The only required method. */
|
||||
/**
|
||||
* Stream one model call as raw chunks. The only required method.
|
||||
* @param options - the fully-assembled request; implementations must honor `options.signal`.
|
||||
* @returns the chunk stream, obeying the adapter contract documented on `StreamChunk`.
|
||||
*/
|
||||
abstract stream(options: GenerateOptions): AsyncIterable<StreamChunk>
|
||||
}
|
||||
|
||||
|
||||
@@ -26,6 +26,9 @@
|
||||
* variant was added without updating the switch (compile error at the call
|
||||
* site — the desired outcome) or a value escaped its type (runtime throw
|
||||
* with diagnostics — the safety net).
|
||||
* @param value - the impossible value; typed `never` so an unhandled variant fails compilation at the call site.
|
||||
* @param context - optional label (e.g. the switch site) prefixed into the throw message.
|
||||
* @returns never — it always throws, with the offending value JSON-rendered in the message.
|
||||
*/
|
||||
export function assertNever(value: never, context?: string): never {
|
||||
// JSON.stringify is typed string but returns undefined for undefined input;
|
||||
|
||||
@@ -70,7 +70,9 @@ export interface ContentBlockMap {
|
||||
'tool-result': ToolResultBlock
|
||||
}
|
||||
|
||||
/** The block `type` tag vocabulary; widens as plugins merge new shapes into {@link ContentBlockMap}. */
|
||||
export type ContentBlockType = keyof ContentBlockMap
|
||||
/** Any known content block, derived from {@link ContentBlockMap}; switch on `type` and fall through unknowns (merge-extensible). */
|
||||
export type ContentBlock = ContentBlockMap[ContentBlockType]
|
||||
|
||||
/** A single message in a conversation history. */
|
||||
@@ -88,6 +90,7 @@ export interface MessageSourceMap {
|
||||
plugin: { kind: 'plugin'; plugin: string }
|
||||
}
|
||||
|
||||
/** Any known message source, derived from {@link MessageSourceMap}; switch on `kind` and fall through unknowns (merge-extensible). */
|
||||
export type MessageSource = MessageSourceMap[keyof MessageSourceMap]
|
||||
|
||||
/**
|
||||
@@ -102,6 +105,7 @@ export interface FinishReasonMap {
|
||||
'error': { kind: 'error'; message: string; code?: string }
|
||||
}
|
||||
|
||||
/** Any known finish reason, derived from {@link FinishReasonMap}; switch on `kind` and fall through unknowns (merge-extensible). */
|
||||
export type FinishReason = FinishReasonMap[keyof FinishReasonMap]
|
||||
|
||||
/**
|
||||
|
||||
@@ -27,7 +27,11 @@ export interface HeaderLine {
|
||||
seedLength?: number
|
||||
}
|
||||
|
||||
/** Build the header line object from a {@link SessionHeader}. */
|
||||
/**
|
||||
* Build the header line object from a {@link SessionHeader}.
|
||||
* @param header - the immutable session metadata to serialize.
|
||||
* @returns the `type: 'session'`-tagged line object, absent optional fields omitted (never null).
|
||||
*/
|
||||
export function toHeaderLine(header: SessionHeader): HeaderLine {
|
||||
return {
|
||||
type: 'session',
|
||||
@@ -40,7 +44,11 @@ export function toHeaderLine(header: SessionHeader): HeaderLine {
|
||||
}
|
||||
}
|
||||
|
||||
/** Parse a header line back into a {@link SessionHeader}. */
|
||||
/**
|
||||
* Parse a header line back into a {@link SessionHeader}.
|
||||
* @param line - the shape-checked first line of a log (see the `isHeaderLine` guard).
|
||||
* @returns the header, absent optional fields omitted.
|
||||
*/
|
||||
export function fromHeaderLine(line: HeaderLine): SessionHeader {
|
||||
return {
|
||||
version: line.version,
|
||||
@@ -77,6 +85,8 @@ function isHeaderLine(value: unknown): value is HeaderLine {
|
||||
* `Buffer.from(…, 'utf8')` would do, breaking injectivity). `.` is in the safe
|
||||
* set for readability but the whole-segment tokens `.`/`..` are escaped so they
|
||||
* can never traverse.
|
||||
* @param raw - the string to encode; must be non-empty (throws on `''`).
|
||||
* @returns the escaped single path segment, decodable back to `raw`.
|
||||
*/
|
||||
export function encodeSegment(raw: string): string {
|
||||
if (raw.length === 0) throw new Error('cannot encode an empty path segment')
|
||||
@@ -97,9 +107,12 @@ export function encodeSegment(raw: string): string {
|
||||
|
||||
/**
|
||||
* The directory a session's files live in: the configured root, then a per-cwd
|
||||
* subdirectory so sessions group by project. The cwd subdir is a stable hash
|
||||
* (short, collision-resistant, filesystem-safe) plus an encoded suffix for
|
||||
* readability; sessions without a cwd go in a shared `_no-cwd` bucket.
|
||||
* subdirectory so sessions group by project. The cwd subdir is a stable hash of
|
||||
* the cwd (short, collision-resistant, filesystem-safe); sessions without a
|
||||
* cwd go in a shared `_no-cwd` bucket.
|
||||
* @param root - the backend's session root directory.
|
||||
* @param cwd - the session's project directory; `undefined` selects the shared `_no-cwd` bucket.
|
||||
* @returns the per-cwd bucket directory path under `root`.
|
||||
*/
|
||||
export function sessionDir(root: string, cwd: string | undefined): string {
|
||||
if (cwd === undefined) return join(root, '_no-cwd')
|
||||
@@ -107,12 +120,22 @@ export function sessionDir(root: string, cwd: string | undefined): string {
|
||||
return join(root, `cwd-${hash}`)
|
||||
}
|
||||
|
||||
/** The append-only event-log file path for a session. */
|
||||
/**
|
||||
* The append-only event-log file path for a session.
|
||||
* @param root - the backend's session root directory.
|
||||
* @param cwd - the session's project directory (picks the per-cwd bucket; `undefined` → `_no-cwd`).
|
||||
* @param id - the session id, path-encoded via {@link encodeSegment} before filesystem use.
|
||||
* @returns the session's `.jsonl` log file path.
|
||||
*/
|
||||
export function logPath(root: string, cwd: string | undefined, id: SessionId): string {
|
||||
return join(sessionDir(root, cwd), `${encodeSegment(id)}.jsonl`)
|
||||
}
|
||||
|
||||
/** Serialize one event as a JSONL line (no trailing newline). */
|
||||
/**
|
||||
* Serialize one event as a JSONL line (no trailing newline).
|
||||
* @param event - the event to serialize verbatim.
|
||||
* @returns the event's single-line JSON text; the writer adds the newline.
|
||||
*/
|
||||
export function eventLine(event: SessionEvent): string {
|
||||
return JSON.stringify(event)
|
||||
}
|
||||
@@ -135,6 +158,9 @@ export function eventLine(event: SessionEvent): string {
|
||||
* This relies on the session-log invariant that every event lives inside a turn
|
||||
* (`Session.append` enforces it): only the final turn can be open, so the
|
||||
* preserved tail is at most one unclosed turn.
|
||||
* @param buffer - the raw bytes of the log file (header line first).
|
||||
* @returns the header, the preserved event prefix, and `committedBytes` — the
|
||||
* byte offset the next append truncates any torn tail to.
|
||||
*/
|
||||
export function scanLog(buffer: Buffer): { meta: SessionHeader; events: SessionEvent[]; committedBytes: number } {
|
||||
const text = buffer.toString('utf8')
|
||||
@@ -239,6 +265,8 @@ export function scanLog(buffer: Buffer): { meta: SessionHeader; events: SessionE
|
||||
* `undefined` if it is missing/not a header. Used by `list()` to read session
|
||||
* metadata WITHOUT parsing the whole log: a session picker scales with the
|
||||
* number of sessions, not the total size of every conversation.
|
||||
* @param firstLine - the first line of a log file (without its trailing newline).
|
||||
* @returns the parsed header, or `undefined` when the line is not a well-formed session header.
|
||||
*/
|
||||
export function parseHeaderMeta(firstLine: string): SessionHeader | undefined {
|
||||
let parsed: unknown
|
||||
|
||||
@@ -31,6 +31,7 @@ import {
|
||||
encodeSegment, eventLine, logPath, parseHeaderMeta, scanLog, sessionDir, toHeaderLine,
|
||||
} from './format.ts'
|
||||
|
||||
/** Plugin config: where the JSONL backend keeps its session logs (`root` is required — no default). */
|
||||
export interface Config {
|
||||
/**
|
||||
* Root directory for all session files. Required (no default): a default of
|
||||
|
||||
@@ -23,6 +23,15 @@ afterEach(async () => {
|
||||
for (const d of dirs.splice(0)) await rm(d, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
function appendClosedTurn(session: Session): void {
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('user/message', {
|
||||
content: [{ type: 'text', text: 'hello' }],
|
||||
source: { kind: 'user' },
|
||||
}, { surfaceOp: 'append' })
|
||||
session.append('turn/end', { turn: 1, reason: { kind: 'completed' } })
|
||||
}
|
||||
|
||||
// Run the shared backend contract against the real JSONL backend.
|
||||
runPersistenceContract('jsonl', async () => {
|
||||
const dir = await mkdtemp(join(tmpdir(), 'dsh-jsonl-'))
|
||||
@@ -131,6 +140,23 @@ describe('SessionPersistenceJsonl: durability and crash semantics', () => {
|
||||
expect(loaded.events).toEqual(log) // chunks preserved, contiguous seqs
|
||||
})
|
||||
|
||||
it('persists a forked child seed through the existing session write path', async () => {
|
||||
const source = ctx.sessions.create(SessionId('persist-parent'), { meta: { cwd: '/workspace' } })
|
||||
appendClosedTurn(source)
|
||||
|
||||
const child = ctx.sessions.fork(source, undefined, SessionId('persist-child'))
|
||||
await ctx.parallel('session/flush', child)
|
||||
const loaded = await ctx.sessionPersistence.load(child.id)
|
||||
|
||||
expect(loaded.events).toEqual(source.events)
|
||||
expect(loaded.meta).toMatchObject({
|
||||
id: SessionId('persist-child'),
|
||||
cwd: '/workspace',
|
||||
parentSession: SessionId('persist-parent'),
|
||||
seedLength: source.events.length,
|
||||
})
|
||||
})
|
||||
|
||||
it('crash recovery: load preserves the interrupted turn and closes it with a synthetic turn/end {interrupted}', async () => {
|
||||
const m = meta('crash', '/proj')
|
||||
await ctx.sessionPersistence.create(m)
|
||||
|
||||
@@ -8,7 +8,7 @@ A SQLite durable session-persistence backend — a second `SessionPersistence` i
|
||||
|
||||
Each `SessionEvent` maps 1:1 onto a row in an `events` table `(session_id, seq, type, time, data, source_event_seqs, surface_op)` — `data` is the event payload as JSON text, so the row shape is the event verbatim (including `assistant/chunk`, keeping `seq` contiguous). The two `TEXT` columns `source_event_seqs` and `surface_op` are nullable; they store the event's optional surface-metadata fields (see [session surface](../../../docs/rfc/implemented/architecture/2026-06-18-session-surface.md)). Out-of-log metadata (`SessionHeader`) lives in a `sessions` row. A `sessions` row is written only by the first `append` — its existence is the lazy-materialization signal (`list` reports exactly the sessions that have a row), so no separate column is needed.
|
||||
|
||||
The repo targets Node ≥ 24 (the root `engines` field), which includes the stable `node:sqlite` module. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software).
|
||||
The repo's `engines.node` is `^22.19.0 || >=24.0.0` (Node 22.19+ or 24+), matching the LTS floor required by the installed Pi adapter dependency; `node:sqlite` itself ships without the `--experimental-sqlite` flag from Node 22.13 (LTS) and 23.4 / 24 (Current) on. The range deliberately excludes Node 23 because that line is non-LTS/EOL and still has flagged runtime features before 23.6. The database opens with `foreign_keys = ON` (so `ON DELETE CASCADE` drops a session's events with its row) and the configured `journal_mode` (default `wal`; pick a rollback-journal mode like `delete` on filesystems where WAL's shared-memory files do not work, e.g. network mounts). The table-layout version is stored in `PRAGMA user_version` and checked on open: a fresh database is stamped with the current `SCHEMA_VERSION`; a database written by any other, incompatible build (a non-current `user_version`, older or newer) is rejected rather than opened against an unknown layout — there is no migration (unreleased software).
|
||||
|
||||
## Contract semantics over rows
|
||||
|
||||
|
||||
@@ -76,6 +76,9 @@ export type JournalMode = 'wal' | 'delete' | 'truncate' | 'persist'
|
||||
* is the merged layout carrying every column; bumping past the collided v3
|
||||
* makes the version check reject both sibling v3 databases instead of opening
|
||||
* one against columns it does not have.
|
||||
* @param path - the SQLite database file to open (created when absent).
|
||||
* @param journalMode - the journal pragma to apply — a closed in-code union, validated by the plugin Config.
|
||||
* @returns the open handle with pragmas applied and both tables ensured.
|
||||
*/
|
||||
export function openDatabase(path: string, journalMode: JournalMode): DatabaseSync {
|
||||
const db = new DatabaseSync(path)
|
||||
@@ -120,7 +123,11 @@ export function openDatabase(path: string, journalMode: JournalMode): DatabaseSy
|
||||
return db
|
||||
}
|
||||
|
||||
/** Reconstruct the {@link SessionHeader} from a `sessions` row. */
|
||||
/**
|
||||
* Reconstruct the {@link SessionHeader} from a `sessions` row.
|
||||
* @param row - the `sessions` table row.
|
||||
* @returns the header, `NULL` columns mapped to omitted optional fields.
|
||||
*/
|
||||
export function rowToMeta(row: SessionRow): SessionHeader {
|
||||
return {
|
||||
version: row.version,
|
||||
@@ -132,7 +139,12 @@ export function rowToMeta(row: SessionRow): SessionHeader {
|
||||
}
|
||||
}
|
||||
|
||||
/** Reconstruct a {@link SessionEvent} from an `events` row (parses `data`). */
|
||||
/**
|
||||
* Reconstruct a {@link SessionEvent} from an `events` row (parses `data`).
|
||||
* @param row - the `events` table row; `data` and the surface columns hold JSON text.
|
||||
* @returns the reconstructed event; throws when a JSON column fails to parse
|
||||
* ({@link scanRows} treats that as a hole, not corruption, in the tail).
|
||||
*/
|
||||
export function rowToEvent(row: EventRow): SessionEvent {
|
||||
// Surface-metadata fields are conditional on the event type in the type
|
||||
// system; spread them so each variant gets only the fields it declares.
|
||||
@@ -172,6 +184,9 @@ export function rowToEvent(row: EventRow): SessionEvent {
|
||||
* This relies on the session-log invariant that every event lives inside a turn
|
||||
* (`Session.append` enforces it): only the final turn can be open, so the
|
||||
* preserved tail is at most one unclosed turn.
|
||||
* @param rows - one session's event rows, ordered by seq ascending.
|
||||
* @returns the preserved event prefix, plus `tornFrom` — the seq the physical
|
||||
* delete starts at — when a torn tail exists.
|
||||
*/
|
||||
export function scanRows(rows: readonly EventRow[]): { preserved: SessionEvent[]; tornFrom?: number } {
|
||||
// Pass 1: parse each row's data; a row whose data is not valid JSON is a hole.
|
||||
|
||||
@@ -186,6 +186,7 @@ export class PersistenceCoordinator<TornMarker = unknown> {
|
||||
/**
|
||||
* Register a new session's metadata (lazy: no physical write until the first
|
||||
* {@link append}). Rejects if the id is already tracked or already persisted.
|
||||
* @param meta - the immutable header (id, version, cwd, lineage) to record; snapshotted at call time.
|
||||
*/
|
||||
create(meta: SessionHeader): Promise<void> {
|
||||
// Snapshot the metadata at call time: the op runs later (behind the
|
||||
@@ -216,6 +217,8 @@ export class PersistenceCoordinator<TornMarker = unknown> {
|
||||
/**
|
||||
* Durably persist a batch of events. Honors the append-only and contiguous-seq
|
||||
* contracts; rejects non-JSON-serializable `event.data`.
|
||||
* @param id - the session the batch belongs to.
|
||||
* @param events - the contiguous batch to persist, in seq order; deep-cloned at call time.
|
||||
*/
|
||||
async append(id: SessionId, events: readonly SessionEvent[]): Promise<void> {
|
||||
// Validate serializability BEFORE cloning so a bad event surfaces the typed
|
||||
@@ -252,6 +255,8 @@ export class PersistenceCoordinator<TornMarker = unknown> {
|
||||
* Reload a session: its {@link SessionHeader} plus the event log up to the last
|
||||
* durable checkpoint, with any interrupted final turn durably closed (synthetic
|
||||
* boundary events) during load.
|
||||
* @param id - the persisted session to reload.
|
||||
* @returns the header plus the event log, ending on a balanced `turn/end`.
|
||||
*/
|
||||
load(id: SessionId): Promise<{ meta: SessionHeader; events: SessionEvent[] }> {
|
||||
return this.serialize(id, () => this.loadCore(id))
|
||||
|
||||
@@ -45,6 +45,9 @@ declare module 'cordis' {
|
||||
*
|
||||
* The comparison includes the full event payload, not just seq/type/time, so a
|
||||
* mutated seed cannot be grafted onto a durable log with the same envelope.
|
||||
* @param seed - the live session's creation-time event snapshot.
|
||||
* @param prefix - the persisted prefix the seed must reproduce.
|
||||
* @returns `true` when the prefix fits within the seed and every event matches by JSON text.
|
||||
*/
|
||||
export function seedCoversPrefix(seed: readonly SessionEvent[], prefix: readonly SessionEvent[]): boolean {
|
||||
return prefix.length <= seed.length
|
||||
@@ -58,6 +61,7 @@ export function seedCoversPrefix(seed: readonly SessionEvent[], prefix: readonly
|
||||
* Reject non-JSON-serializable event data before a backend serializes a batch.
|
||||
* Live session appends already enforce this; persistence append paths also
|
||||
* accept replay/fork batches that may bypass a live session instance.
|
||||
* @param events - the batch to validate; throws naming the offending event's type and seq.
|
||||
*/
|
||||
export function assertSerializable(events: readonly SessionEvent[]): void {
|
||||
for (const event of events) {
|
||||
|
||||
@@ -121,7 +121,12 @@ export const DEFAULT_DISPOSE_GRACE_MS = 3_000
|
||||
*/
|
||||
export const SENSITIVE_ENV_PATTERN = /KEY|SECRET|TOKEN/i
|
||||
|
||||
/** The ambient env minus credential-shaped vars, plus the spec's explicit env. */
|
||||
/**
|
||||
* The ambient env minus credential-shaped vars, plus the spec's explicit env.
|
||||
* @param extra - explicit vars layered on top AFTER the scrub, so a
|
||||
* credential-shaped name supplied deliberately still reaches the child.
|
||||
* @returns the environment to spawn the child with.
|
||||
*/
|
||||
export function buildChildEnv(extra: Record<string, string>): NodeJS.ProcessEnv {
|
||||
const env: NodeJS.ProcessEnv = {}
|
||||
for (const [key, value] of Object.entries(process.env)) {
|
||||
@@ -130,7 +135,12 @@ export function buildChildEnv(extra: Record<string, string>): NodeJS.ProcessEnv
|
||||
return { ...env, ...extra }
|
||||
}
|
||||
|
||||
/** Map an ACP {@link StopReason} to a harness {@link SubagentStopReason}. */
|
||||
/**
|
||||
* Map an ACP {@link StopReason} to a harness {@link SubagentStopReason}.
|
||||
* @param reason - the terminal reason from the child's `session/prompt` response.
|
||||
* @returns the harness equivalent; `max_turn_requests` and any unknown future
|
||||
* variant map to `error`, so an unclean stop is never reported as `completed`.
|
||||
*/
|
||||
export function acpStopReason(reason: StopReason): SubagentStopReason {
|
||||
switch (reason) {
|
||||
case 'end_turn':
|
||||
@@ -155,12 +165,20 @@ export function acpStopReason(reason: StopReason): SubagentStopReason {
|
||||
}
|
||||
}
|
||||
|
||||
/** Collect the text of an ACP content block (non-text blocks contribute nothing). */
|
||||
/**
|
||||
* Collect the text of an ACP content block (non-text blocks contribute nothing).
|
||||
* @param content - the content block off a streamed `agent_message_chunk`.
|
||||
* @returns the block's text, or `''` for a non-text block.
|
||||
*/
|
||||
export function acpContentText(content: AcpContentBlock): string {
|
||||
return content.type === 'text' ? content.text : ''
|
||||
}
|
||||
|
||||
/** Translate the harness prompt blocks into ACP prompt blocks (text only). */
|
||||
/**
|
||||
* Translate the harness prompt blocks into ACP prompt blocks (text only).
|
||||
* @param prompt - the harness prompt; non-text blocks are dropped.
|
||||
* @returns the ACP text blocks, in order.
|
||||
*/
|
||||
export function toAcpPrompt(prompt: ContentBlock[]): AcpContentBlock[] {
|
||||
const blocks: AcpContentBlock[] = []
|
||||
for (const block of prompt) {
|
||||
@@ -206,6 +224,11 @@ function exitsWithin(child: ChildProcess, ms: number): Promise<boolean> {
|
||||
* failure (a spawn/transport/RPC error resolves with `stopReason: 'error'`), per
|
||||
* the seam contract. `cancel()` sends `session/cancel`; `dispose()` kills the
|
||||
* subprocess and awaits its exit (quiescent teardown).
|
||||
* @param request - the start request; the driver consumes `prompt` and `signal`
|
||||
* (an already-aborted signal yields an inert `aborted` run with no spawn).
|
||||
* @param spec - the resolved spawn spec: command/args/cwd, env, permission
|
||||
* policy, dispose graces, and the optional error sink.
|
||||
* @returns the live run handle for the child subprocess.
|
||||
*/
|
||||
export function startAcpRun(request: SubagentStartRequest, spec: AcpRunSpec): SubagentRun {
|
||||
const id = AgentId(randomUUID())
|
||||
|
||||
@@ -12,7 +12,7 @@ The seam this rides on: `CreateAgentOptions.seed` (added on `dsh-agent`, threade
|
||||
|
||||
## Capabilities
|
||||
|
||||
`{ outputSchema: false, depthLimit: true, toolFilter: false }` — identical to spawn (the depth/model/output behavior is the shared driver's).
|
||||
`{ outputSchema: true, depthLimit: true, toolFilter: false }` — identical to spawn (the depth/model/structured-output behavior is the shared driver's).
|
||||
|
||||
## Config
|
||||
|
||||
|
||||
@@ -28,6 +28,10 @@ import type { SubagentCapabilities, SubagentProvider, SubagentStartRequest } fro
|
||||
import { startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess'
|
||||
|
||||
export const name = 'subagent-fork'
|
||||
// `tools` is deliberately NOT injected — same rationale as subagent-spawn: the
|
||||
// per-run structured runtime gates its capture-tool registration on `tools`
|
||||
// itself, so this backend's apply timing (and the delegation tool's position
|
||||
// in the model-visible tool list) is unchanged by structured output.
|
||||
export const inject = ['subagents', 'agents']
|
||||
|
||||
/** Config: the registry name to register the provider under. */
|
||||
@@ -47,6 +51,8 @@ export const Config: z<Config> = z.object({
|
||||
* empty — i.e. fresh — child). The result is contiguous from seq 0 (the live
|
||||
* log keeps `seq === index`), so it is a valid session seed; the in-flight,
|
||||
* unbalanced turn is dropped so the invariants replay accepts it.
|
||||
* @param parent - the agent whose session log to slice.
|
||||
* @returns the seed events, contiguous from seq 0; empty when no turn has completed.
|
||||
*/
|
||||
export function completedTurnPrefix(parent: Agent): SessionEvent[] {
|
||||
const events = parent.session.events
|
||||
@@ -57,11 +63,12 @@ export function completedTurnPrefix(parent: Agent): SessionEvent[] {
|
||||
}
|
||||
|
||||
/**
|
||||
* The fork provider. Supports `depthLimit`; NOT `outputSchema`/`toolFilter` this
|
||||
* cut (the service rejects a request needing either before `start` runs).
|
||||
* The fork provider. Supports `depthLimit` and `outputSchema` (via the shared
|
||||
* in-process structured runtime); NOT `toolFilter` this cut (the service
|
||||
* rejects a request needing it before `start` runs).
|
||||
*/
|
||||
class ForkProvider implements SubagentProvider {
|
||||
readonly capabilities: SubagentCapabilities = { outputSchema: false, depthLimit: true, toolFilter: false }
|
||||
readonly capabilities: SubagentCapabilities = { outputSchema: true, depthLimit: true, toolFilter: false }
|
||||
// Context contract: a forked child IS seeded with the parent's completed-turn prefix.
|
||||
readonly inheritsParentContext = true
|
||||
|
||||
|
||||
@@ -9,9 +9,10 @@ import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
|
||||
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
|
||||
import * as Invariants from '@deepseek-ai/dsh-invariants'
|
||||
import SubagentService from '@deepseek-ai/dsh-subagent'
|
||||
import { MockAdapter, textResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
|
||||
import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
|
||||
import type { StreamChunk } from '@deepseek-ai/dsh-llm'
|
||||
import * as fork from '../src/index.ts'
|
||||
import { STRUCTURED_OUTPUT_TOOL } from '@deepseek-ai/dsh-subagent-inprocess'
|
||||
import { completedTurnPrefix } from '../src/index.ts'
|
||||
|
||||
type Script = ConstructorParameters<typeof MockAdapter>[0]
|
||||
@@ -141,6 +142,26 @@ describe('dsh-subagent-fork', () => {
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('captures structured output through the shipped plugin (seeded child, driver runtime)', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
textResponse('parent turn'),
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }),
|
||||
])
|
||||
parent.send([{ type: 'text', text: 'warm up' }])
|
||||
await parent.whenIdle()
|
||||
const run = ctx.subagents.start('fork', {
|
||||
prompt: [{ type: 'text', text: 'report structured' }],
|
||||
parent,
|
||||
outputSchema: { type: 'object', properties: { answer: { type: 'number' } }, required: ['answer'] },
|
||||
})
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('completed')
|
||||
expect(result.structured).toEqual({ answer: 9 })
|
||||
// Run-scoped runtime: nothing stays registered after the settle.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('does NOT return the seeded parent output when the child produces no message of its own', async () => {
|
||||
// Regression: readResult must scope to the child's OWN events (after the
|
||||
// seed). The parent completes a turn with a distinctive assistant message,
|
||||
@@ -161,9 +182,9 @@ describe('dsh-subagent-fork', () => {
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('advertises depthLimit but not outputSchema/toolFilter', async () => {
|
||||
it('advertises depthLimit and outputSchema but not toolFilter', async () => {
|
||||
const { ctx } = await setup([])
|
||||
expect(ctx.subagents.getProvider('fork')!.capabilities).toEqual({ outputSchema: false, depthLimit: true, toolFilter: false })
|
||||
expect(ctx.subagents.getProvider('fork')!.capabilities).toEqual({ outputSchema: true, depthLimit: true, toolFilter: false })
|
||||
})
|
||||
|
||||
it('unregisters the provider when its fiber is disposed (HMR safety)', async () => {
|
||||
|
||||
@@ -8,10 +8,10 @@ The shared **in-process subagent run driver**. A pure library (no provider, no r
|
||||
|
||||
Runs a child as a child [`Agent`](../../core/agent) on the same cordis context (`ctx.agents`):
|
||||
|
||||
1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability);
|
||||
2. creates a child via `ctx.agents.create` with a fresh `AgentId`/`SessionId`, the parent's `cwd` + `parentSession` lineage, the optional `options.seed` (fork's completed-turn prefix; omitted for a fresh child), and `agentOptions` (the child inherits the **parent's model** by default — a child with no model can't run — overridable via `request.agentOptions.model`; the system prompt is NOT inherited);
|
||||
3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts);
|
||||
4. reads the result, scoped to the child's OWN events (everything at or after `seedLength`, so a seeded child that produced no message of its own never returns the seeded parent's last message): the last `assistant/message` content (deep-cloned — the log is frozen) and the last `turn/end.reason` mapped to a `SubagentStopReason`.
|
||||
1. computes child depth = `depthOf(parent) + 1`; if `request.maxDepth` is set and exceeded, throws `SubagentDepthError` (the `depthLimit` capability); a `request.outputSchema` is asserted against the supported subset (`assertSupportedOutputSchema` from [dsh-tools](../../core/tools/README.md)) and then snapshotted with `structuredClone` before any child exists — assertion first so a hostile value fails as `OutputSchemaError` (never a raw clone error), the snapshot so a post-`start()` caller mutation cannot drift the enforced schema;
|
||||
2. creates a child via `ctx.agents.create` with a fresh `AgentId`/`SessionId`, the parent's `cwd` + `parentSession` lineage, the optional `options.seed` (fork's completed-turn prefix; omitted for a fresh child), and `agentOptions` (the child inherits the **parent's model** by default — a child with no model can't run — overridable via `request.agentOptions.model`; the deployment persona needs no inheritance — it is a context-wide prompt section);
|
||||
3. drives the one-shot: `child.send(prompt)` then `await child.whenIdle()` (ordering matters — `send` enqueues synchronously, so `whenIdle` observes the queued work and resolves on the child's `running → idle` transition, never before the turn starts); there is deliberately NO re-prompt for a structured child that finished cleanly without calling `structured_output` — the shortfall maps to an `error` result for the parent;
|
||||
4. reads the result, scoped to the child's OWN events (everything at or after `seedLength`, so a seeded child that produced no message of its own never returns the seeded parent's last message): the last `assistant/message` content (deep-cloned — the log is frozen) and the last `turn/end.reason` mapped to a `SubagentStopReason`. A structured run surfaces the captured value as `result.structured`; a structured child that finished cleanly WITHOUT ever capturing settles `error` (a clean finish without the demanded result is a failure, not a success with a missing field).
|
||||
|
||||
`dispose()` delegates to `AgentHandle.dispose()` (stop loop → await quiescence → remove session); `cancel()` cancels the child's in-flight turn. A cancel landing before any `turn/end` (the pre-turn window) still settles `aborted`, honoring the cancel contract rather than the generic no-turn `error`.
|
||||
|
||||
@@ -19,6 +19,19 @@ Runs a child as a child [`Agent`](../../core/agent) on the same cordis context (
|
||||
|
||||
`{ providerName: string; seed?: SessionEvent[] }` — the per-backend inputs: the provider name (for error context) and the optional child-session seed.
|
||||
|
||||
### Structured output (package-internal runtime)
|
||||
|
||||
The mechanism behind `outputSchema` for in-process children — acquired per structured RUN inside `startInProcessRun` (nothing is registered on a context that never runs a structured child; only the model-facing constants `STRUCTURED_OUTPUT_TOOL`/`STRUCTURED_OUTPUT_INSTRUCTION` are exported). One globally registered `structured_output` capture tool (its registered parameters are a placeholder) plus four listeners:
|
||||
|
||||
- a `system-prompt/assemble` waterfall listener registered `prepend: true` that post-processes `await next()` — **final-assembly enforcement**: the assembly the loop renders never carries `structured_output` for an agent without a structured run, and for one that has it always carries the run's OWN schema (as the tool's `parameters`) plus the calling instruction as a trailing prompt section (the demand travels with the tool — `AgentOptions` has no per-agent prompt field to carry it). The loop logs the rendered assembly as the step's `request/header`, so the injection is reconstructable log state, never a wire-only mutation. Per-agent shaping lives here because the tool registry and prompt assembly are context-global while schemas differ per concurrent child (FIXME in the module doc: per-agent/per-session scoping would dissolve this); cooperative mutate-then-`next()` would not survive a downstream listener returning a replacement assembly.
|
||||
- a `tools/post-execute` listener (`prepend: true` = outermost, so `await next()` yields the composed final decision) that COMMITS the capture: the tool body only stages the validated value, and it becomes the run's result only when the final decision accepts the call — a downstream block (a PostToolUse hook) turns the logged result into `isError`, and the run must not report `structured` success for a call the model and session log saw fail.
|
||||
- a `tools/pre-execute` deny for any call arriving after the agent's capture — terminal means terminal WITHIN the step: a response listing `structured_output` before further tool calls cannot run side effects after the final answer was accepted.
|
||||
- an `agent/turn-continuation` listener (also `prepend: true` — an earlier-registered force-continue listener returning without `next()` must not decide the turn before the veto runs) that stops a child's turn once its output is captured, so a successful capture doesn't buy a wasted extra model step.
|
||||
|
||||
The capture tool validates each call against the run's schema (`validateStructuredValue`) — violations become an `INVALID_ARGS` isError result the model retries in-turn; a valid call stages the value for the post-execute commit.
|
||||
|
||||
Lifetime is refcounted by live structured runs: each acquires at start and releases at settle, so the registrations exist exactly while at least one structured child is live, a backend hot-reload mid-run cannot unregister the capture tool under a live child, and the last settle disposes everything. `release()` is idempotent per acquisition.
|
||||
|
||||
### `depthOf(agent): number`
|
||||
|
||||
Delegation depth rides on a merge-extensible `AgentOptions.subagentDepth` field (0 for a top-level agent, parent + 1 for a child), so a nested spawn reads its parent's depth from `parent.options.subagentDepth`. `depthOf` reads it (absent ⇒ 0).
|
||||
|
||||
@@ -26,6 +26,8 @@
|
||||
"@deepseek-ai/dsh-llm": "^0.0.1",
|
||||
"@deepseek-ai/dsh-session": "^0.0.1",
|
||||
"@deepseek-ai/dsh-subagent": "^0.0.1",
|
||||
"@deepseek-ai/dsh-system-prompt": "^0.0.1",
|
||||
"@deepseek-ai/dsh-tools": "^0.0.1",
|
||||
"cordis": "^4.0.0-rc.6"
|
||||
},
|
||||
"devDependencies": {
|
||||
|
||||
@@ -18,7 +18,20 @@ import type { Context } from 'cordis'
|
||||
import { AgentId, type Agent, type AgentHandle, type AgentOptions } from '@deepseek-ai/dsh-agent'
|
||||
import { SessionId, type SessionEvent, type TurnEndReason } from '@deepseek-ai/dsh-session'
|
||||
import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
||||
import { assertSupportedOutputSchema } from '@deepseek-ai/dsh-tools'
|
||||
import type { SubagentResult, SubagentRun, SubagentStartRequest, SubagentStopReason } from '@deepseek-ai/dsh-subagent'
|
||||
import {
|
||||
acquireStructuredRuntime,
|
||||
type StructuredAcquisition,
|
||||
} from './structured.ts'
|
||||
|
||||
// The runtime itself (acquire/attach/release) is package-internal: runs
|
||||
// acquire it inside startInProcessRun, and no other package drives it. Only
|
||||
// the model-facing vocabulary is public.
|
||||
export {
|
||||
STRUCTURED_OUTPUT_TOOL,
|
||||
STRUCTURED_OUTPUT_INSTRUCTION,
|
||||
} from './structured.ts'
|
||||
|
||||
declare module '@deepseek-ai/dsh-agent' {
|
||||
interface AgentOptions {
|
||||
@@ -34,7 +47,11 @@ declare module '@deepseek-ai/dsh-agent' {
|
||||
}
|
||||
}
|
||||
|
||||
/** Read an agent's delegation depth (absent ⇒ a top-level agent, depth 0). */
|
||||
/**
|
||||
* Read an agent's delegation depth (absent ⇒ a top-level agent, depth 0).
|
||||
* @param agent - the agent whose options may carry `subagentDepth`.
|
||||
* @returns 0 for a top-level agent, its parent's depth + 1 for a subagent.
|
||||
*/
|
||||
export function depthOf(agent: Agent): number {
|
||||
return agent.options.subagentDepth ?? 0
|
||||
}
|
||||
@@ -88,6 +105,13 @@ export interface InProcessRunOptions {
|
||||
* the matching `turn/end.reason` the stop reason. `dispose()` delegates to the
|
||||
* factory's {@link AgentHandle.dispose} (stop loop → await quiescence → remove
|
||||
* session); `cancel()` cancels the child's in-flight turn.
|
||||
*
|
||||
* Throws {@link SubagentDepthError} before creating anything when the child's
|
||||
* depth (parent depth + 1) would exceed `request.maxDepth`.
|
||||
* @param ctx - the context whose `agents` factory creates and owns the child.
|
||||
* @param request - the start request (prompt, parent, signal, per-child options).
|
||||
* @param options - the backend's inputs: provider name plus the optional seed.
|
||||
* @returns the live run handle for the child agent.
|
||||
*/
|
||||
export function startInProcessRun(
|
||||
ctx: Context,
|
||||
@@ -98,6 +122,18 @@ export function startInProcessRun(
|
||||
if (request.maxDepth !== undefined && childDepth > request.maxDepth) {
|
||||
throw new SubagentDepthError(childDepth, request.maxDepth)
|
||||
}
|
||||
// Assert, then snapshot, the schema subset BEFORE any child exists (the
|
||||
// service has already capability-gated; this rejects a schema outside the
|
||||
// enforced subset loud). Assertion comes FIRST so a hostile value fails as
|
||||
// OutputSchemaError, never as structuredClone's raw DataCloneError — the
|
||||
// asserted subset is plain JSON data, which always clones. The snapshot is
|
||||
// load-bearing: the caller keeps its reference, so attaching the ORIGINAL
|
||||
// would let a post-start() mutation drift the enforced schema away from the
|
||||
// asserted one — the clone (taken synchronously with the assertion, no
|
||||
// interleaving possible) pins assertion, the model-visible parameters, and
|
||||
// validateStructuredValue to one isolation-immutable value.
|
||||
if (request.outputSchema !== undefined) assertSupportedOutputSchema(request.outputSchema)
|
||||
const schema = request.outputSchema === undefined ? undefined : structuredClone(request.outputSchema)
|
||||
|
||||
const childId = AgentId(randomUUID())
|
||||
// The child's OWN events begin after the seed (fork seeds the parent's
|
||||
@@ -109,13 +145,20 @@ export function startInProcessRun(
|
||||
// Inherit the parent's model by default (a child with no model cannot run);
|
||||
// an explicit `request.agentOptions.model` overrides it. The persona needs
|
||||
// no inheritance: the deployment persona is a context-wide prompt section,
|
||||
// so parent and child render the same one.
|
||||
// so parent and child render the same one. A structured run's
|
||||
// structured_output instruction is NOT prompt state either — the structured
|
||||
// runtime's final-request listener appends it per request (see structured.ts).
|
||||
const agentOptions: AgentOptions = {
|
||||
...request.parent.options.model !== undefined ? { model: request.parent.options.model } : {},
|
||||
...request.agentOptions,
|
||||
subagentDepth: childDepth,
|
||||
}
|
||||
|
||||
// The structured runtime is held for the WHOLE run (acquired before the child
|
||||
// exists, released when the result settles), so a backend hot-reload mid-run
|
||||
// cannot unregister the capture tool out from under this live child.
|
||||
const structured: StructuredAcquisition | undefined = schema !== undefined ? acquireStructuredRuntime(ctx) : undefined
|
||||
|
||||
const handle: AgentHandle = ctx.agents.create({
|
||||
agentId: childId,
|
||||
sessionId: SessionId(randomUUID()),
|
||||
@@ -130,6 +173,7 @@ export function startInProcessRun(
|
||||
agentOptions,
|
||||
})
|
||||
const child = handle.agent
|
||||
if (structured && schema !== undefined) structured.attach(child, schema)
|
||||
|
||||
// Bridge the request's abort signal to the child (the consumer also bridges
|
||||
// its own exec.signal, but a backend-level bridge keeps the contract local).
|
||||
@@ -138,6 +182,10 @@ export function startInProcessRun(
|
||||
// `turn/end` is logged — settles as `aborted` (honoring the cancel contract)
|
||||
// rather than falling through to the no-turn `error` mapping.
|
||||
let cancelled = false
|
||||
// An accessor, not an inline read: `cancelled` mutates from closures (the
|
||||
// abort listener, run.cancel), which control-flow narrowing cannot see — an
|
||||
// inline read at the result mapping would narrow to the initializer.
|
||||
const isCancelled = (): boolean => cancelled
|
||||
const requestCancel = (reason: string): void => {
|
||||
cancelled = true
|
||||
child.cancel(reason)
|
||||
@@ -154,9 +202,16 @@ export function startInProcessRun(
|
||||
if (request.signal?.aborted) return { output: [], stopReason: 'aborted' }
|
||||
child.send(request.prompt)
|
||||
await child.whenIdle()
|
||||
return readResult(child, seedLength, cancelled)
|
||||
// Deliberately NO re-prompt when a structured child finishes cleanly
|
||||
// without calling structured_output: readResult maps that to `error` —
|
||||
// the shortfall goes to the parent instead of buying extra model turns.
|
||||
return readResult(child, seedLength, isCancelled(), structured ? { captured: structured.captured(child) } : undefined)
|
||||
} finally {
|
||||
request.signal?.removeEventListener('abort', onAbort)
|
||||
if (structured) {
|
||||
structured.detach(child)
|
||||
structured.release()
|
||||
}
|
||||
}
|
||||
})()
|
||||
|
||||
@@ -184,12 +239,32 @@ export function startInProcessRun(
|
||||
* logged (a cancel landed in the pre-turn window, before any turn ran), the
|
||||
* run settles `aborted` per the {@link SubagentRun.cancel} contract rather than
|
||||
* the generic no-turn `error`.
|
||||
*
|
||||
* A structured run (`structured` present) additionally reports the captured
|
||||
* value on {@link SubagentResult.structured}. A structured child that finished
|
||||
* CLEANLY without ever capturing (the nudges ran out) settles `error` — a clean
|
||||
* finish without the demanded structured result is a failure, not a success
|
||||
* with a missing field; a non-`completed` reason keeps its own honest mapping.
|
||||
*/
|
||||
function readResult(child: Agent, seedLength: number, cancelled: boolean): SubagentResult {
|
||||
function readResult(
|
||||
child: Agent,
|
||||
seedLength: number,
|
||||
cancelled: boolean,
|
||||
structured?: { captured?: { value: unknown } | undefined },
|
||||
): SubagentResult {
|
||||
const own = child.session.events.slice(seedLength)
|
||||
const lastMessage = own.findLast((e): e is SessionEvent<'assistant/message'> => e.type === 'assistant/message')
|
||||
const lastEnd = own.findLast((e): e is SessionEvent<'turn/end'> => e.type === 'turn/end')
|
||||
const output: ContentBlock[] = lastMessage ? structuredClone(lastMessage.data.content) : []
|
||||
if (lastEnd === undefined && cancelled) return { output, stopReason: 'aborted' }
|
||||
return { output, stopReason: toStopReason(lastEnd?.data.reason) }
|
||||
const stopReason: SubagentStopReason = lastEnd === undefined && cancelled
|
||||
? 'aborted'
|
||||
: toStopReason(lastEnd?.data.reason)
|
||||
if (structured) {
|
||||
if (structured.captured) return { output, structured: structured.captured.value, stopReason }
|
||||
// No capture on a cleanly-completed turn: an ERROR when the run was left
|
||||
// to finish (the nudges ran out), but ABORTED when a cancel is why the
|
||||
// nudging stopped — the cancel contract outranks the schema shortfall.
|
||||
if (stopReason === 'completed') return { output, stopReason: cancelled ? 'aborted' : 'error' }
|
||||
}
|
||||
return { output, stopReason }
|
||||
}
|
||||
|
||||
312
packages/subagent/subagent-inprocess/src/structured.ts
Normal file
312
packages/subagent/subagent-inprocess/src/structured.ts
Normal file
@@ -0,0 +1,312 @@
|
||||
/**
|
||||
* Structured-output support for the in-process subagent backends: the mechanism
|
||||
* behind `SubagentStartRequest.outputSchema` for children that run as agents on
|
||||
* the same context.
|
||||
*
|
||||
* The model-facing surface is one globally registered `structured_output` tool
|
||||
* whose REGISTERED parameters are a placeholder — the real schema is per run.
|
||||
* Because the tool registry and prompt assembly are context-global while
|
||||
* schemas differ per child (two concurrent structured runs may carry different
|
||||
* schemas), per-agent shaping happens on the `system-prompt/assemble`
|
||||
* waterfall with a `prepend: true` listener that post-processes `await next()`
|
||||
* — FINAL-ASSEMBLY enforcement: whatever downstream listeners mutated or
|
||||
* replaced, the assembly the loop renders never carries `structured_output`
|
||||
* for an agent without a structured run, and for one that has it always
|
||||
* carries the run's OWN schema plus a trailing
|
||||
* {@link STRUCTURED_OUTPUT_INSTRUCTION} section (the demand travels with the
|
||||
* tool). The loop logs what the assembly produced as the request header, so
|
||||
* the injection is a reconstructable fact of the session log, never a
|
||||
* wire-only mutation (the reconstructability RFC).
|
||||
* (Cooperative mutate-then-`next()` would not survive a downstream listener
|
||||
* returning a replacement assembly — see the waterfall composition caveat in
|
||||
* docs/architecture.md.)
|
||||
*
|
||||
* FIXME: the whole enforcement dance above exists because the tool registry
|
||||
* and prompt assembly are context-global. If they become per-agent or
|
||||
* per-session scoped, a structured run just registers its own schema'd tool on
|
||||
* the child's scope and this module reduces to the capture tool plus the
|
||||
* turn-stop — no placeholder, no final-assembly swap, no strip-for-everyone-
|
||||
* else, no global-registration lifetime dance.
|
||||
*
|
||||
* A companion `agent/turn-continuation` listener stops a child's turn once its
|
||||
* output is captured — without it, the loop's default "had tool calls ⇒
|
||||
* continue" buys a wasted extra model step per structured child. It is also
|
||||
* `prepend: true`: the veto must run before any earlier-registered listener
|
||||
* that could short-circuit the chain into a forced continue. A third listener
|
||||
* closes the within-step window the continuation veto cannot: a
|
||||
* `tools/pre-execute` deny for any call arriving after the agent's capture, so
|
||||
* a response that lists `structured_output` before further tool calls cannot
|
||||
* run side effects after the final answer was accepted. A fourth,
|
||||
* `tools/post-execute`, is the capture COMMIT: the tool body only stages the
|
||||
* validated value, and it becomes the run's captured result only when the
|
||||
* final post-execute decision accepts the call — a blocking hook downstream
|
||||
* yields `isError` in the log, and the run must not report success for it.
|
||||
*
|
||||
* Lifetime is refcounted by structured RUNS: each acquires from start to
|
||||
* settle, so the registrations exist exactly while at least one structured
|
||||
* child is live — a plain deployment that never passes `outputSchema` carries
|
||||
* no always-on global state, and a backend hot-reload mid-run cannot
|
||||
* unregister the capture tool out from under a live child (the run holds its
|
||||
* own acquisition). Registrations land on the ROOT context and the refcount
|
||||
* disposes them when the last run settles; the next structured run
|
||||
* re-registers them.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-subagent-inprocess/structured
|
||||
*/
|
||||
|
||||
import type { Context } from 'cordis'
|
||||
import type { Agent } from '@deepseek-ai/dsh-agent'
|
||||
import type { ContentBlock, ToolSchema } from '@deepseek-ai/dsh-llm'
|
||||
import type { ContinuationDecision } from '@deepseek-ai/dsh-agent'
|
||||
import type { AssembleContext, PromptAssembly } from '@deepseek-ai/dsh-system-prompt'
|
||||
import type { PostToolDecision, PreToolDecision, ToolExecution, ToolExecutionResult } from '@deepseek-ai/dsh-tools'
|
||||
import { ToolArgsError, validateStructuredValue, type StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
|
||||
|
||||
/** The model-facing tool name a structured child must call to finish. */
|
||||
export const STRUCTURED_OUTPUT_TOOL = 'structured_output'
|
||||
|
||||
/**
|
||||
* The instruction the assembly listener appends to a structured child's
|
||||
* system prompt as a trailing section on every assembly. Per-assembly state,
|
||||
* NOT agent prompt state: `AgentOptions` has no prompt field (the persona is
|
||||
* deployment config on the system-prompt plugin), so the same final-assembly
|
||||
* enforcement that injects the schema'd tool carries the instruction that
|
||||
* demands calling it.
|
||||
*/
|
||||
export const STRUCTURED_OUTPUT_INSTRUCTION
|
||||
= 'When you have your final answer, you MUST report it by calling the '
|
||||
+ `\`${STRUCTURED_OUTPUT_TOOL}\` tool with arguments matching its parameter schema exactly. `
|
||||
+ 'Do not finish with a plain text answer: only the tool call counts as your result.'
|
||||
|
||||
/** One structured run's state: the schema to enforce and the captured value, once recorded. */
|
||||
interface RunState {
|
||||
readonly schema: StructuredOutputSchema
|
||||
/**
|
||||
* A validated value awaiting the post-execute verdict on ITS OWN call. Set
|
||||
* by the capture tool's body, promoted to {@link RunState.captured} only
|
||||
* when the final `tools/post-execute` decision accepts the call — a
|
||||
* downstream block turns the logged result into `isError`, and a value
|
||||
* committed at body time would let the run report success for a call the
|
||||
* model saw fail.
|
||||
*/
|
||||
pending?: { value: unknown }
|
||||
captured?: { value: unknown }
|
||||
}
|
||||
|
||||
/** The per-root-context runtime: run states plus the shared registrations. */
|
||||
interface StructuredRuntime {
|
||||
refs: number
|
||||
readonly states: WeakMap<Agent, RunState>
|
||||
readonly disposers: (() => void)[]
|
||||
}
|
||||
|
||||
/** One root context ⇒ one runtime (multi-app test isolation). */
|
||||
const runtimes = new WeakMap<Context, StructuredRuntime>()
|
||||
|
||||
/**
|
||||
* One holder's handle on the shared structured runtime. `release()` is
|
||||
* idempotent per acquisition; the runtime's registrations are disposed when the
|
||||
* LAST holder (backend plugin or live run) releases.
|
||||
*/
|
||||
export interface StructuredAcquisition {
|
||||
/** Enforce `schema` on `agent`'s requests and start capturing its `structured_output` call. */
|
||||
attach(agent: Agent, schema: StructuredOutputSchema): void
|
||||
/** The captured value, once the child called the tool with valid arguments. */
|
||||
captured(agent: Agent): { value: unknown } | undefined
|
||||
/** Stop enforcing/capturing for `agent` (WeakMap-backed; safe to call twice). */
|
||||
detach(agent: Agent): void
|
||||
/** Drop this holder's reference (idempotent); the last release unregisters everything. */
|
||||
release(): void
|
||||
}
|
||||
|
||||
/**
|
||||
* Acquire the per-root-context structured runtime, registering the capture tool
|
||||
* and the runtime's listeners on the FIRST acquisition. See the module doc
|
||||
* for the enforcement and lifetime design.
|
||||
* @param ctx - any context of the app; the runtime keys off `ctx.root`.
|
||||
* @returns this holder's handle (attach/captured/detach + idempotent release).
|
||||
*/
|
||||
export function acquireStructuredRuntime(ctx: Context): StructuredAcquisition {
|
||||
const root: Context = ctx.root
|
||||
let runtime = runtimes.get(root)
|
||||
if (!runtime) {
|
||||
runtime = { refs: 0, states: new WeakMap(), disposers: [] }
|
||||
runtimes.set(root, runtime)
|
||||
registerRuntime(root, runtime)
|
||||
}
|
||||
runtime.refs += 1
|
||||
|
||||
let released = false
|
||||
return {
|
||||
attach(agent: Agent, schema: StructuredOutputSchema): void {
|
||||
runtime.states.set(agent, { schema })
|
||||
},
|
||||
captured(agent: Agent): { value: unknown } | undefined {
|
||||
return runtime.states.get(agent)?.captured
|
||||
},
|
||||
detach(agent: Agent): void {
|
||||
runtime.states.delete(agent)
|
||||
},
|
||||
release(): void {
|
||||
if (released) return
|
||||
released = true
|
||||
runtime.refs -= 1
|
||||
if (runtime.refs > 0) return
|
||||
runtimes.delete(root)
|
||||
for (const dispose of runtime.disposers.splice(0)) dispose()
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/** Register the capture tool + the two listeners on the root context (first acquire). */
|
||||
function registerRuntime(root: Context, runtime: StructuredRuntime): void {
|
||||
// The registered parameters are a PLACEHOLDER: the request listener below
|
||||
// swaps in the run's real schema per child, and strips the tool entirely for
|
||||
// every agent without a structured run — so this shape is never model-visible.
|
||||
//
|
||||
// Registration does NOT ride on the acquiring backend's plugin-level
|
||||
// `inject`: a backend that waited on `tools` would apply later than it did
|
||||
// before this module existed, shifting when its PROVIDER registers — and the
|
||||
// delegation tool mirrors provider lifecycle, so that shift would reorder
|
||||
// the model-visible tool list of every existing prompt. Instead the capture
|
||||
// tool registers synchronously when `tools` is already live (the common
|
||||
// case), and through a scoped inject fiber when the Loader happens to start
|
||||
// the backend first. Either way the registration lands on root and is
|
||||
// disposed by the runtime's refcount; disposing the fiber also covers the
|
||||
// never-activated case.
|
||||
let disposeTool: (() => void) | undefined
|
||||
const registerCapture = (tools: Context['tools']): void => {
|
||||
disposeTool = tools.register({
|
||||
name: STRUCTURED_OUTPUT_TOOL,
|
||||
description:
|
||||
'Report your final structured result. Call this exactly once, when your answer is complete; '
|
||||
+ 'the arguments must match this tool\'s parameter schema exactly.',
|
||||
parameters: { type: 'object', properties: {} },
|
||||
execute(args: unknown, exec: ToolExecution): Promise<ContentBlock[]> {
|
||||
const state = exec.agent ? runtime.states.get(exec.agent) : undefined
|
||||
if (!state) {
|
||||
// Reachable only if a non-structured agent somehow calls the tool (the
|
||||
// request listener strips it, so the model never sees it) — fail loud
|
||||
// rather than capture into nowhere.
|
||||
throw new Error(`${STRUCTURED_OUTPUT_TOOL} is only available to subagents started with an output schema`)
|
||||
}
|
||||
const violations = validateStructuredValue(state.schema, args)
|
||||
// ToolArgsError → isError result with INVALID_ARGS: the model retries
|
||||
// within the same turn, exactly like a schema-validated defineTool call.
|
||||
if (violations.length > 0) throw new ToolArgsError(violations)
|
||||
// Two-phase commit: the body only STAGES the value; the post-execute
|
||||
// listener below promotes it once the final decision accepts the call.
|
||||
state.pending = { value: args }
|
||||
return Promise.resolve([{ type: 'text', text: 'Structured output recorded.' }])
|
||||
},
|
||||
})
|
||||
}
|
||||
const liveTools = root.get('tools')
|
||||
const toolsFiber = liveTools ? undefined : root.inject(['tools'], (childCtx: Context) => {
|
||||
registerCapture(childCtx.root.tools)
|
||||
})
|
||||
if (liveTools) registerCapture(liveTools)
|
||||
runtime.disposers.push(() => {
|
||||
disposeTool?.()
|
||||
void toolsFiber?.dispose()
|
||||
})
|
||||
|
||||
// FINAL-ASSEMBLY enforcement (prepend: true = first registered = OUTERMOST
|
||||
// wrapper): post-process whatever the downstream listeners and the registry
|
||||
// produced, so a downstream listener returning a replacement assembly cannot
|
||||
// leak the tool to other agents or erase the child's schema. The loop logs
|
||||
// the rendered assembly as the step's request header, so the swap is
|
||||
// reconstructable log state, never a wire-only mutation.
|
||||
runtime.disposers.push(root.on('system-prompt/assemble', async function (
|
||||
this: unknown, _assembly: PromptAssembly, context: AssembleContext, next: () => Promise<PromptAssembly>,
|
||||
): Promise<PromptAssembly> {
|
||||
const final = await next()
|
||||
const state = context.agent ? runtime.states.get(context.agent) : undefined
|
||||
if (state) {
|
||||
const schemaEntry: ToolSchema = {
|
||||
name: STRUCTURED_OUTPUT_TOOL,
|
||||
description:
|
||||
'Report your final structured result. Call this exactly once, when your answer is complete; '
|
||||
+ 'the arguments must match this tool\'s parameter schema exactly.',
|
||||
// ToolSchema.parameters is the wire-level JSON Schema object; the
|
||||
// asserted subset type is structurally exactly that.
|
||||
parameters: state.schema as unknown as Record<string, unknown>,
|
||||
}
|
||||
final.tools = [...final.tools.filter(tool => tool.name !== STRUCTURED_OUTPUT_TOOL), schemaEntry]
|
||||
// The demand travels WITH the tool: a trailing section in the
|
||||
// tool-guidance order band, appended after next() so it renders last
|
||||
// (renderPrompt joins in array order).
|
||||
final.sections = [...final.sections, { name: `tool:${STRUCTURED_OUTPUT_TOOL}`, order: 190, text: STRUCTURED_OUTPUT_INSTRUCTION }]
|
||||
return final
|
||||
}
|
||||
// No structured run: strip the placeholder so it is never model-visible.
|
||||
// An empty tools array canonicalizes to an absent header/wire field
|
||||
// (canonicalHeader pins empty ≡ absent), so no re-shaping is needed here.
|
||||
final.tools = final.tools.filter(tool => tool.name !== STRUCTURED_OUTPUT_TOOL)
|
||||
return final
|
||||
}, { prepend: true }))
|
||||
|
||||
// Stop a structured child's turn once its output is captured: the default
|
||||
// "had tool calls ⇒ continue" would otherwise buy a wasted extra model step
|
||||
// after every successful capture. `prepend: true` puts the veto OUTERMOST —
|
||||
// an earlier-registered listener that short-circuits the chain (a goal-style
|
||||
// force-continue returning without `next()`) would otherwise decide the turn
|
||||
// before this listener ever ran, and no downstream decision may resurrect a
|
||||
// structured turn that is already finished.
|
||||
runtime.disposers.push(root.on('agent/turn-continuation', function (
|
||||
this: unknown, agent: Agent, _turn: number, _decision: ContinuationDecision, next: () => Promise<ContinuationDecision>,
|
||||
): Promise<ContinuationDecision> {
|
||||
if (runtime.states.get(agent)?.captured) return Promise.resolve({ action: 'stop' })
|
||||
return next()
|
||||
}, { prepend: true }))
|
||||
|
||||
// The capture COMMIT: promote the staged value only when the final
|
||||
// post-execute decision accepts the call. The capture tool's body cannot
|
||||
// decide — `tools/post-execute` runs after it, and a blocking listener (a
|
||||
// PostToolUse hook) turns the logged result into `isError` feedback; a value
|
||||
// committed at body time would make readResult report `structured` success
|
||||
// for a call whose result the model and session log saw fail. `prepend:
|
||||
// true` = outermost at registration time, so `await next()` returns the
|
||||
// COMPOSED downstream decision — the same final verdict the registry maps
|
||||
// onto the result. (A later-registered outer listener that blocks without
|
||||
// delegating skips this commit entirely: the staged value is dropped and the
|
||||
// run errors — failure-safe in the same direction.) The staging slot clears
|
||||
// on every path, including a rejecting downstream listener.
|
||||
runtime.disposers.push(root.on('tools/post-execute', async function (
|
||||
this: unknown, exec: ToolExecution, _result: ToolExecutionResult, next: () => Promise<PostToolDecision>,
|
||||
): Promise<PostToolDecision> {
|
||||
const state = exec.agent ? runtime.states.get(exec.agent) : undefined
|
||||
if (!state || exec.name !== STRUCTURED_OUTPUT_TOOL || state.pending === undefined) return next()
|
||||
const pending = state.pending
|
||||
try {
|
||||
const decision = await next()
|
||||
if (decision.kind === 'accept') state.captured = pending
|
||||
return decision
|
||||
} finally {
|
||||
delete state.pending
|
||||
}
|
||||
}, { prepend: true }))
|
||||
|
||||
// Terminal means terminal WITHIN the step, not only at its end: the
|
||||
// turn-continuation veto above runs after every call in the current model
|
||||
// response has executed, so a response that puts `structured_output` before
|
||||
// further tool calls would still perform those side effects after the final
|
||||
// answer was accepted. Deny every later call for a captured agent at the
|
||||
// allow/deny gate — dispatch is skipped and the model sees an `isError`
|
||||
// result naming the contract. Calls that PRECEDE the capture in the same
|
||||
// response ran before `captured` was set and are untouched; a second
|
||||
// `structured_output` is denied like any other call. `prepend: true` for the
|
||||
// same reason as the continuation veto: no earlier-registered allow may
|
||||
// short-circuit past the terminal contract.
|
||||
runtime.disposers.push(root.on('tools/pre-execute', function (
|
||||
this: unknown, exec: ToolExecution, next: () => Promise<PreToolDecision>,
|
||||
): Promise<PreToolDecision> {
|
||||
if (exec.agent && runtime.states.get(exec.agent)?.captured) {
|
||||
return Promise.resolve({
|
||||
kind: 'deny',
|
||||
reason: `structured output already recorded: the run is complete, so \`${exec.name}\` is not executed`,
|
||||
})
|
||||
}
|
||||
return next()
|
||||
}, { prepend: true }))
|
||||
}
|
||||
606
packages/subagent/subagent-inprocess/tests/structured.spec.ts
Normal file
606
packages/subagent/subagent-inprocess/tests/structured.spec.ts
Normal file
@@ -0,0 +1,606 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService, { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm'
|
||||
import SessionStore from '@deepseek-ai/dsh-session'
|
||||
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
|
||||
import ToolRegistry from '@deepseek-ai/dsh-tools'
|
||||
import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
|
||||
import type { Agent, ContinuationDecision } from '@deepseek-ai/dsh-agent'
|
||||
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
|
||||
import * as Invariants from '@deepseek-ai/dsh-invariants'
|
||||
import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent'
|
||||
import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
|
||||
import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
|
||||
import { startInProcessRun } from '../src/index.ts'
|
||||
import {
|
||||
acquireStructuredRuntime,
|
||||
STRUCTURED_OUTPUT_INSTRUCTION,
|
||||
STRUCTURED_OUTPUT_TOOL,
|
||||
} from '../src/structured.ts'
|
||||
|
||||
type Script = ConstructorParameters<typeof MockAdapter>[0]
|
||||
|
||||
const SCHEMA: StructuredOutputSchema = {
|
||||
type: 'object',
|
||||
properties: { answer: { type: 'number' }, note: { type: 'string' } },
|
||||
required: ['answer'],
|
||||
}
|
||||
|
||||
/**
|
||||
* Real loop + scripted mock model + an INLINE spawn-shaped provider over the
|
||||
* shared driver. The concrete backend plugins are deliberately NOT loaded —
|
||||
* they would devDep-cycle this package (spawn/fork already depend on the
|
||||
* driver), and the runtime under test is the driver's; plugin-level structured
|
||||
* coverage lives in the spawn/fork specs. The mock model script drives the
|
||||
* child's structured_output calls.
|
||||
*/
|
||||
async function setup(script: Script) {
|
||||
const ctx = new Context()
|
||||
const adapter = new MockAdapter(script)
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await ctx.plugin(AgentRegistry)
|
||||
await ctx.plugin(Invariants)
|
||||
await ctx.plugin(AgentLoop, { agents: [] })
|
||||
await ctx.plugin(SubagentService)
|
||||
const disposeProvider = ctx.subagents.registerProvider({
|
||||
name: 'spawn',
|
||||
capabilities: { outputSchema: true, depthLimit: true, toolFilter: false },
|
||||
inheritsParentContext: false,
|
||||
start: (request: SubagentStartRequest) => startInProcessRun(ctx, request, { providerName: 'spawn' }),
|
||||
})
|
||||
ctx.llm.registerAdapter(['mock'], adapter)
|
||||
const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' })
|
||||
return { ctx, parent, adapter, disposeProvider }
|
||||
}
|
||||
|
||||
function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial<SubagentStartRequest>): SubagentStartRequest {
|
||||
return { prompt: [{ type: 'text', text: 'produce the answer' }], parent, outputSchema: SCHEMA, ...extra }
|
||||
}
|
||||
|
||||
/** The tool names of one recorded model request. */
|
||||
function toolNames(request: GenerateOptions): string[] {
|
||||
return (request.tools ?? []).map(tool => tool.name)
|
||||
}
|
||||
|
||||
describe('in-process structured output', () => {
|
||||
it('captures a valid structured_output call and surfaces result.structured', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }),
|
||||
])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('completed')
|
||||
expect(result.structured).toEqual({ answer: 42, note: 'done' })
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('stops the turn after a successful capture — no extra model step is spent', async () => {
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
|
||||
textResponse('MUST NOT BE CONSUMED'),
|
||||
])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
await run.result
|
||||
// Default continuation would run a second step after the tool call; the
|
||||
// structured runtime's turn-continuation veto stops the turn instead.
|
||||
expect(adapter.requests.length).toBe(1)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => {
|
||||
// One model response carrying structured_output FIRST and a side-effecting
|
||||
// call after it: the continuation veto only fires at step end, so without
|
||||
// the pre-execute deny the trailing call would still run after the final
|
||||
// answer was accepted.
|
||||
const response = [
|
||||
...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
|
||||
{ type: 'block-start', index: 1, blockType: 'tool-call' },
|
||||
{ type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
|
||||
{ type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
|
||||
{ type: 'finish', reason: { kind: 'tool-calls' } },
|
||||
] as Script[number]
|
||||
const { ctx, parent } = await setup([response])
|
||||
let sideEffectRan = false
|
||||
ctx.tools.register({
|
||||
name: 'side_effect',
|
||||
description: 'probe',
|
||||
parameters: { type: 'object', properties: {} },
|
||||
execute(): Promise<ContentBlock[]> {
|
||||
sideEffectRan = true
|
||||
return Promise.resolve([{ type: 'text', text: 'ran' }])
|
||||
},
|
||||
})
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('completed')
|
||||
expect(result.structured).toEqual({ answer: 5 })
|
||||
// The deny skipped dispatch entirely: the probe body never ran.
|
||||
expect(sideEffectRan).toBe(false)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => {
|
||||
const response = [
|
||||
{ type: 'block-start', index: 0, blockType: 'tool-call' },
|
||||
{ type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'side_effect', arguments: '{}' } },
|
||||
...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk =>
|
||||
'index' in chunk ? { ...chunk, index: 1 } : chunk),
|
||||
] as Script[number]
|
||||
const { ctx, parent } = await setup([response])
|
||||
let sideEffectRan = false
|
||||
ctx.tools.register({
|
||||
name: 'side_effect',
|
||||
description: 'probe',
|
||||
parameters: { type: 'object', properties: {} },
|
||||
execute(): Promise<ContentBlock[]> {
|
||||
sideEffectRan = true
|
||||
return Promise.resolve([{ type: 'text', text: 'ran' }])
|
||||
},
|
||||
})
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
// The call ran BEFORE captured was set: the deny gate only guards the
|
||||
// window after the terminal answer landed.
|
||||
expect(sideEffectRan).toBe(true)
|
||||
expect(result.structured).toEqual({ answer: 6 })
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('snapshots the schema at start(): caller mutation after start cannot drift enforcement', async () => {
|
||||
const mutable: StructuredOutputSchema = {
|
||||
type: 'object',
|
||||
properties: { answer: { type: 'number' } },
|
||||
required: ['answer'],
|
||||
additionalProperties: false,
|
||||
}
|
||||
const pristine = structuredClone(mutable)
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }),
|
||||
])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: mutable }))
|
||||
// Mutate the caller's object AFTER start() returned but before the child's
|
||||
// first request assembles: with a live reference this would reach both the
|
||||
// model-visible parameters and validateStructuredValue.
|
||||
;(mutable.properties as Record<string, unknown>).answer = { type: 'string' }
|
||||
const result = await run.result
|
||||
expect(result.structured).toEqual({ answer: 3 })
|
||||
// The child's request carried the PRISTINE schema, not the mutated one.
|
||||
const childRequest = adapter.requests.at(-1)
|
||||
const captureTool = (childRequest?.tools ?? []).find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
|
||||
expect(captureTool?.parameters).toEqual(pristine)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('the captured-turn veto is prepend: an EARLIER force-continue listener cannot short-circuit it', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
// Registered BEFORE the structured runtime exists — without prepend, this
|
||||
// goal-style listener would decide the turn first (returning WITHOUT
|
||||
// calling next()) and the veto would never run.
|
||||
ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'continue' }))
|
||||
const acquisition = acquireStructuredRuntime(ctx)
|
||||
const agent = { id: AgentId('structured-child') } as unknown as Agent
|
||||
acquisition.attach(agent, SCHEMA)
|
||||
const captured = await ctx.tools.execute({
|
||||
callId: 'call-1' as never,
|
||||
name: STRUCTURED_OUTPUT_TOOL,
|
||||
arguments: { answer: 1 },
|
||||
agent,
|
||||
})
|
||||
expect(captured.isError).toBeFalsy()
|
||||
const decision = await ctx.waterfall(
|
||||
'agent/turn-continuation', agent, 1,
|
||||
{ action: 'continue' },
|
||||
() => Promise.resolve<ContinuationDecision>({ action: 'continue' }),
|
||||
)
|
||||
expect(decision).toEqual({ action: 'stop' })
|
||||
acquisition.detach(agent)
|
||||
acquisition.release()
|
||||
})
|
||||
|
||||
it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }),
|
||||
toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
|
||||
])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.structured).toEqual({ answer: 7 })
|
||||
expect(result.stopReason).toBe('completed')
|
||||
// The child's log carries the isError tool/result for the invalid call.
|
||||
const child = ctx.agents.get(run.id)!
|
||||
const results = child.session.events.filter(e => e.type === 'tool/result')
|
||||
expect(results.length).toBe(2)
|
||||
expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => {
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
textResponse('here is my answer in prose'),
|
||||
textResponse('MUST NOT BE CONSUMED'),
|
||||
])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('error')
|
||||
expect(result.structured).toBeUndefined()
|
||||
// Exactly one model request and one user message: no nudge turn exists.
|
||||
expect(adapter.requests.length).toBe(1)
|
||||
const child = ctx.agents.get(run.id)!
|
||||
expect(child.session.events.filter(e => e.type === 'user/message').length).toBe(1)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('an errored child keeps its honest error result (no capture expected)', async () => {
|
||||
// Script exhaustion on the first call → the child turn errors.
|
||||
const { ctx, parent, adapter } = await setup([])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('error')
|
||||
expect(adapter.requests.length).toBe(1)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => {
|
||||
const { ctx, parent } = await setup([textResponse('prose, no capture')])
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const child = ctx.agents.get(run.id)!
|
||||
// Cancel synchronously inside the turn's end recording: the cancel
|
||||
// contract outranks the schema shortfall, so the result maps to aborted.
|
||||
ctx.on('session/event', (session, event) => {
|
||||
if (session === child.session && event.type === 'turn/end') run.cancel('cancelled at turn end')
|
||||
})
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('aborted')
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('rejects a schema outside the subset loud, before any child exists', async () => {
|
||||
const { ctx, parent } = await setup([])
|
||||
expect(() => ctx.subagents.start('spawn', structuredRequest(parent, {
|
||||
outputSchema: { type: 'object', oneOf: [] } as unknown as StructuredOutputSchema,
|
||||
}))).toThrow(/unsupported output schema/)
|
||||
expect(ctx.agents.get(AgentId('parent'))).toBeDefined()
|
||||
})
|
||||
|
||||
it('a schema carrying non-JSON values fails as OutputSchemaError, never as a raw clone error', async () => {
|
||||
const { ctx, parent } = await setup([])
|
||||
// Assertion runs BEFORE the defensive structuredClone: a function-valued
|
||||
// annotation must surface as the subset violation it is, not escape as
|
||||
// structuredClone's DataCloneError.
|
||||
expect(() => ctx.subagents.start('spawn', structuredRequest(parent, {
|
||||
outputSchema: { type: 'object', default: () => {} } as unknown as StructuredOutputSchema,
|
||||
}))).toThrow(/unsupported output schema.*annotation must be JSON data/)
|
||||
})
|
||||
|
||||
it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => {
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
|
||||
textResponse('continues after the blocked capture'),
|
||||
])
|
||||
// A PostToolUse-style hook, registered AFTER the runtime (so the runtime's
|
||||
// prepend commit listener stays outermost and composes this verdict).
|
||||
ctx.on('tools/post-execute', (exec, _result, next) => {
|
||||
if (exec.name === STRUCTURED_OUTPUT_TOOL) {
|
||||
return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] })
|
||||
}
|
||||
return next()
|
||||
})
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
// No capture was committed: the run reports the schema shortfall...
|
||||
expect(result.structured).toBeUndefined()
|
||||
expect(result.stopReason).toBe('error')
|
||||
// ...the logged tool result is the blocked isError with the feedback...
|
||||
const child = ctx.agents.get(run.id)!
|
||||
const results = child.session.events.filter(e => e.type === 'tool/result')
|
||||
expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
|
||||
expect(JSON.stringify((results[0]!.data as { content: unknown }).content)).toContain('capture rejected by hook')
|
||||
// ...and the turn CONTINUED past the blocked call (no captured veto):
|
||||
// the model got to react to the failure with a second step.
|
||||
expect(adapter.requests.length).toBe(2)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('a post-execute accept-with-replacement still commits the capture', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
|
||||
])
|
||||
ctx.on('tools/post-execute', (exec, _result, next) => {
|
||||
if (exec.name === STRUCTURED_OUTPUT_TOOL) {
|
||||
return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] })
|
||||
}
|
||||
return next()
|
||||
})
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('completed')
|
||||
expect(result.structured).toEqual({ answer: 8 })
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => {
|
||||
const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })])
|
||||
// A context-wide section stands in for the deployment persona: the
|
||||
// instruction must APPEND to whatever the prompt pipeline assembled, not
|
||||
// replace it (AgentOptions has no prompt field — the instruction is
|
||||
// per-request wire state added by the final-request listener).
|
||||
ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' })
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
await run.result
|
||||
const childRequest = adapter.requests.at(-1)!
|
||||
expect(childRequest.system).toContain('You are a counter.')
|
||||
expect(childRequest.system!.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
|
||||
expect(childRequest.system!.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => {
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
textResponse('parent answer'),
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
|
||||
])
|
||||
parent.send([{ type: 'text', text: 'hello' }])
|
||||
await parent.whenIdle()
|
||||
expect(adapter.requests[0]!.system ?? '').not.toContain(STRUCTURED_OUTPUT_INSTRUCTION)
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
await run.result
|
||||
// The loop always assembles a base prompt (the harness identity section),
|
||||
// so the instruction APPENDS — never replaces.
|
||||
const childSystem = adapter.requests.at(-1)!.system!
|
||||
expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
|
||||
expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
describe('final-request enforcement (the prepend agent/request listener)', () => {
|
||||
it('a plain agent assembling while the runtime is LIVE gets the placeholder stripped', async () => {
|
||||
// Run-scoped acquisition means a plain deployment never registers the
|
||||
// tool at all; the strip branch exists for the CONCURRENT case — a plain
|
||||
// agent taking a turn while some structured child holds the runtime open.
|
||||
const { ctx, parent, adapter } = await setup([textResponse('parent answer')])
|
||||
const hold = acquireStructuredRuntime(ctx)
|
||||
parent.send([{ type: 'text', text: 'hello' }])
|
||||
await parent.whenIdle()
|
||||
// The placeholder IS in the registry during this turn; the assembly the
|
||||
// loop rendered must not carry it for an agent without a structured run.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined()
|
||||
expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
|
||||
hold.release()
|
||||
})
|
||||
|
||||
it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => {
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
// Parent turn (a plain agent): must NOT see the tool.
|
||||
textResponse('parent answer'),
|
||||
// Child turn: must see it, with the run's schema.
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
|
||||
])
|
||||
parent.send([{ type: 'text', text: 'hello' }])
|
||||
await parent.whenIdle()
|
||||
expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
|
||||
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
await run.result
|
||||
const childRequest = adapter.requests[1]!
|
||||
expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL)
|
||||
const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
|
||||
expect(entry.parameters).toEqual(SCHEMA)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('two concurrent structured children each see their OWN schema', async () => {
|
||||
const otherSchema: StructuredOutputSchema = {
|
||||
type: 'object',
|
||||
properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } },
|
||||
required: ['verdict'],
|
||||
}
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
(options: GenerateOptions) => {
|
||||
// Answer with whatever schema this child was given — proves each
|
||||
// request carried the right one regardless of scheduling order.
|
||||
const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
|
||||
const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
|
||||
? { verdict: 'real' }
|
||||
: { answer: 1 }
|
||||
return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args)
|
||||
},
|
||||
(options: GenerateOptions) => {
|
||||
const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
|
||||
const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
|
||||
? { verdict: 'real' }
|
||||
: { answer: 1 }
|
||||
return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args)
|
||||
},
|
||||
])
|
||||
const runA = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const runB = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema }))
|
||||
const [a, b] = await Promise.all([runA.result, runB.result])
|
||||
expect(a.structured).toEqual({ answer: 1 })
|
||||
expect(b.structured).toEqual({ verdict: 'real' })
|
||||
const schemas = adapter.requests.map(request =>
|
||||
request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters)
|
||||
expect(schemas).toContainEqual(SCHEMA)
|
||||
expect(schemas).toContainEqual(otherSchema)
|
||||
await runA.dispose()
|
||||
await runB.dispose()
|
||||
})
|
||||
|
||||
it('wins against a downstream listener that REPLACES the assembly object', async () => {
|
||||
const { ctx, parent, adapter } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }),
|
||||
])
|
||||
// A downstream (non-prepend) listener that returns a brand-new assembly —
|
||||
// the composition caveat that erases cooperative mutations. Registered
|
||||
// AFTER the runtime's prepend listener, so it runs INSIDE it.
|
||||
ctx.on('system-prompt/assemble', async (_assembly, _context, next) => {
|
||||
const replaced = await next()
|
||||
return { sections: [...replaced.sections], tools: [...replaced.tools], variables: { ...replaced.variables } }
|
||||
})
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
expect(result.structured).toEqual({ answer: 5 })
|
||||
const entry = adapter.requests[0]!.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
|
||||
expect(entry).toBeDefined()
|
||||
expect(entry!.parameters).toEqual(SCHEMA)
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => {
|
||||
const { parent, adapter } = await setup([
|
||||
// The registry contributes the placeholder via prompt assembly, so
|
||||
// tools is an array in the raw request — but after stripping the
|
||||
// placeholder (its ONLY entry), the field must not be re-added as a
|
||||
// different shape.
|
||||
textResponse('plain'),
|
||||
])
|
||||
parent.send([{ type: 'text', text: 'q' }])
|
||||
await parent.whenIdle()
|
||||
const request = adapter.requests[0]!
|
||||
expect(toolNames(request)).not.toContain(STRUCTURED_OUTPUT_TOOL)
|
||||
await new Promise(resolve => setTimeout(resolve, 0))
|
||||
})
|
||||
|
||||
it('shapes a bare assembly on the waterfall: no-agent context strips the placeholder; a structured agent gains schema + trailing instruction section', async () => {
|
||||
// Drive ctx.systemPrompt.assemble directly — the enforcement listener
|
||||
// must tolerate a context with NO agent (a bare diagnostic assemble)
|
||||
// and shape a structured agent's assembly on the same path the loop
|
||||
// renders and logs as the request header.
|
||||
const { ctx, parent } = await setup([])
|
||||
const acquisition = acquireStructuredRuntime(ctx)
|
||||
// Bare assemble WHILE the runtime is live: the no-agent branch must
|
||||
// strip the registered placeholder (before the acquisition there is
|
||||
// nothing to strip — run-scoped registration).
|
||||
const bare = await ctx.systemPrompt.assemble({})
|
||||
expect(bare.tools.map(tool => tool.name)).not.toContain(STRUCTURED_OUTPUT_TOOL)
|
||||
|
||||
acquisition.attach(parent, SCHEMA)
|
||||
const shaped = await ctx.systemPrompt.assemble({ agent: parent })
|
||||
expect(shaped.tools.map(tool => tool.name)).toContain(STRUCTURED_OUTPUT_TOOL)
|
||||
expect(shaped.tools.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters).toEqual(SCHEMA)
|
||||
// The demand travels with the tool: the instruction renders LAST
|
||||
// (appended post-next(); renderPrompt joins in array order).
|
||||
expect(shaped.sections.at(-1)).toMatchObject({ name: `tool:${STRUCTURED_OUTPUT_TOOL}`, text: STRUCTURED_OUTPUT_INSTRUCTION })
|
||||
acquisition.detach(parent)
|
||||
acquisition.release()
|
||||
})
|
||||
})
|
||||
|
||||
describe('runtime lifetime (refcount: live structured runs)', () => {
|
||||
it('the runtime exists exactly while structured runs are live: nothing before, nothing after', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }),
|
||||
])
|
||||
// No always-on global state: a context that has run no structured child
|
||||
// carries no capture tool.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
const run = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const result = await run.result
|
||||
// The capture succeeded — the registrations existed while the run lived.
|
||||
expect(result.structured).toEqual({ answer: 4 })
|
||||
// The run's settle released the last acquisition.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('concurrent structured runs share one runtime; the last settle disposes it', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
|
||||
toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 2 }),
|
||||
])
|
||||
const first = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const second = ctx.subagents.start('spawn', structuredRequest(parent))
|
||||
const [a, b] = await Promise.all([first.result, second.result])
|
||||
expect([a.structured, b.structured].sort()).toEqual([{ answer: 1 }, { answer: 2 }].sort())
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
await first.dispose()
|
||||
await second.dispose()
|
||||
})
|
||||
|
||||
it('acquisition release is idempotent (double release cannot underflow the refcount)', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
const first = acquireStructuredRuntime(ctx)
|
||||
const second = acquireStructuredRuntime(ctx)
|
||||
first.release()
|
||||
first.release()
|
||||
// The second holder still keeps the tool registered.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined()
|
||||
second.release()
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('registers the capture tool through the scoped fiber when tools loads after the acquisition', async () => {
|
||||
// The Loader starts sibling plugins concurrently, so a backend can
|
||||
// acquire the runtime before dsh-tools has applied. The capture tool
|
||||
// must then register as soon as `tools` exists — via the inject fiber,
|
||||
// not by deferring the backend (which would reorder the prompt's tools).
|
||||
const ctx = new Context()
|
||||
const acquisition = acquireStructuredRuntime(ctx)
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
// Fiber activation completes asynchronously after the service appears.
|
||||
await new Promise(resolve => setImmediate(resolve))
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined()
|
||||
acquisition.release()
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('releasing before tools ever loads disposes the pending fiber without registering', async () => {
|
||||
const ctx = new Context()
|
||||
const acquisition = acquireStructuredRuntime(ctx)
|
||||
acquisition.release()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await new Promise(resolve => setImmediate(resolve))
|
||||
// The disposed fiber never fires: nothing registers after the fact.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('attach/captured/detach manage per-agent state through the acquisition surface', async () => {
|
||||
const { ctx, parent } = await setup([])
|
||||
const acquisition = acquireStructuredRuntime(ctx)
|
||||
expect(acquisition.captured(parent)).toBeUndefined()
|
||||
acquisition.attach(parent, SCHEMA)
|
||||
expect(acquisition.captured(parent)).toBeUndefined()
|
||||
acquisition.detach(parent)
|
||||
acquisition.detach(parent)
|
||||
acquisition.release()
|
||||
// That manual acquisition was the ONLY holder - release disposes.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
it('a direct structured_output call from an agent WITHOUT a structured run is an isError', async () => {
|
||||
const { ctx, parent } = await setup([])
|
||||
// Hold the runtime open (run-scoped: nothing is registered otherwise) so
|
||||
// the call reaches the capture tool's own fail-loud guard, not UNKNOWN_TOOL.
|
||||
const hold = acquireStructuredRuntime(ctx)
|
||||
const result = await ctx.tools.execute({
|
||||
callId: 'x' as never,
|
||||
name: STRUCTURED_OUTPUT_TOOL,
|
||||
arguments: { answer: 1 },
|
||||
agent: parent,
|
||||
})
|
||||
expect(result.isError).toBe(true)
|
||||
expect(JSON.stringify(result.content)).toContain('only available to subagents')
|
||||
hold.release()
|
||||
})
|
||||
|
||||
it('a structured_output call with NO calling agent at all is an isError', async () => {
|
||||
const { ctx } = await setup([])
|
||||
const hold = acquireStructuredRuntime(ctx)
|
||||
const result = await ctx.tools.execute({
|
||||
callId: 'x' as never,
|
||||
name: STRUCTURED_OUTPUT_TOOL,
|
||||
arguments: { answer: 1 },
|
||||
})
|
||||
expect(result.isError).toBe(true)
|
||||
hold.release()
|
||||
})
|
||||
})
|
||||
@@ -25,6 +25,12 @@
|
||||
},
|
||||
{
|
||||
"path": "../subagent"
|
||||
},
|
||||
{
|
||||
"path": "../../core/system-prompt"
|
||||
},
|
||||
{
|
||||
"path": "../../core/tools"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -10,7 +10,7 @@ The run mechanics live in the shared [`@deepseek-ai/dsh-subagent-inprocess`](../
|
||||
|
||||
## Capabilities
|
||||
|
||||
`{ outputSchema: false, depthLimit: true, toolFilter: false }`. It constructs the child, so it enforces a recursion cap; structured output and tool-scoping are deferred (the service rejects a request needing either before `start` runs).
|
||||
`{ outputSchema: true, depthLimit: true, toolFilter: false }`. It constructs the child, so it enforces a recursion cap, and it supports structured output via the driver's [structured runtime](../subagent-inprocess/README.md) (acquired per structured run inside the driver — this backend registers nothing at apply). Tool-scoping is deferred (the service rejects a request needing it before `start` runs).
|
||||
|
||||
## Config
|
||||
|
||||
|
||||
@@ -9,6 +9,11 @@
|
||||
* ({@link startInProcessRun}); this backend just passes NO seed (a fresh
|
||||
* child). The fork backend is an independent peer over the same driver.
|
||||
*
|
||||
* Structured output (`outputSchema`) is supported via the driver's shared
|
||||
* structured runtime: the backend acquires it for its plugin lifetime (so the
|
||||
* capture tool and request-shaping listeners exist before any run), and each
|
||||
* structured run holds its own acquisition until it settles.
|
||||
*
|
||||
* Plugin export shape: named `name`/`inject`/`Config`/`apply`, NO default.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-subagent-spawn
|
||||
@@ -20,6 +25,11 @@ import type { SubagentCapabilities, SubagentProvider, SubagentStartRequest } fro
|
||||
import { startInProcessRun } from '@deepseek-ai/dsh-subagent-inprocess'
|
||||
|
||||
export const name = 'subagent-spawn'
|
||||
// `tools` is deliberately NOT injected: the shared driver's structured runtime
|
||||
// (acquired per structured RUN, not at apply) gates its own capture-tool
|
||||
// registration on `tools` availability, so this backend's apply timing — and
|
||||
// with it the provider-mirroring delegation tool's position in the
|
||||
// model-visible tool list — stays what it was before structured output existed.
|
||||
export const inject = ['subagents', 'agents']
|
||||
|
||||
/** Config: the registry name to register the provider under. */
|
||||
@@ -34,11 +44,12 @@ export const Config: z<Config> = z.object({
|
||||
|
||||
/**
|
||||
* The spawn provider. Supports `depthLimit` (it constructs the child, so it can
|
||||
* enforce a recursion cap) but NOT `outputSchema` or `toolFilter` in this cut —
|
||||
* a request that needs either is rejected by the service before `start` runs.
|
||||
* enforce a recursion cap) and `outputSchema` (via the shared in-process
|
||||
* structured runtime); NOT `toolFilter` in this cut — a request that needs it
|
||||
* is rejected by the service before `start` runs.
|
||||
*/
|
||||
class SpawnProvider implements SubagentProvider {
|
||||
readonly capabilities: SubagentCapabilities = { outputSchema: false, depthLimit: true, toolFilter: false }
|
||||
readonly capabilities: SubagentCapabilities = { outputSchema: true, depthLimit: true, toolFilter: false }
|
||||
// Context contract: a spawned child starts fresh — it never sees the parent conversation.
|
||||
readonly inheritsParentContext = false
|
||||
|
||||
@@ -46,7 +57,8 @@ class SpawnProvider implements SubagentProvider {
|
||||
|
||||
start(request: SubagentStartRequest) {
|
||||
// Fresh child: no seed. The shared driver mints ids, stamps cwd/lineage/
|
||||
// depth, drives the one-shot, and maps the result.
|
||||
// depth, drives the one-shot (including the structured capture when the
|
||||
// request carries an outputSchema), and maps the result.
|
||||
return startInProcessRun(this.ctx, request, { providerName: this.name })
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,9 +10,9 @@ import { SessionId } from '@deepseek-ai/dsh-session'
|
||||
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
|
||||
import * as Invariants from '@deepseek-ai/dsh-invariants'
|
||||
import SubagentService from '@deepseek-ai/dsh-subagent'
|
||||
import { MockAdapter, maxTokensResponse, textResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
|
||||
import { MockAdapter, maxTokensResponse, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
|
||||
import * as spawn from '../src/index.ts'
|
||||
import { depthOf, SubagentDepthError } from '@deepseek-ai/dsh-subagent-inprocess'
|
||||
import { depthOf, STRUCTURED_OUTPUT_TOOL, SubagentDepthError } from '@deepseek-ai/dsh-subagent-inprocess'
|
||||
|
||||
type Script = ConstructorParameters<typeof MockAdapter>[0]
|
||||
|
||||
@@ -241,10 +241,10 @@ describe('dsh-subagent-spawn', () => {
|
||||
await parentHandle.dispose()
|
||||
})
|
||||
|
||||
it('advertises depthLimit but not outputSchema/toolFilter', async () => {
|
||||
it('advertises depthLimit and outputSchema but not toolFilter', async () => {
|
||||
const { ctx } = await setup([])
|
||||
const provider = ctx.subagents.getProvider('spawn')!
|
||||
expect(provider.capabilities).toEqual({ outputSchema: false, depthLimit: true, toolFilter: false })
|
||||
expect(provider.capabilities).toEqual({ outputSchema: true, depthLimit: true, toolFilter: false })
|
||||
})
|
||||
|
||||
it('unregisters the provider when its fiber is disposed (HMR safety)', async () => {
|
||||
@@ -257,6 +257,54 @@ describe('dsh-subagent-spawn', () => {
|
||||
expect(ctx.subagents.list()).toEqual([])
|
||||
})
|
||||
|
||||
it('captures structured output through the shipped plugin (driver runtime, plugin wiring)', async () => {
|
||||
const { ctx, parent } = await setup([
|
||||
toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
|
||||
])
|
||||
const run = ctx.subagents.start('spawn', {
|
||||
prompt: [{ type: 'text', text: 'produce the answer' }],
|
||||
parent,
|
||||
outputSchema: { type: 'object', properties: { answer: { type: 'number' } }, required: ['answer'] },
|
||||
})
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('completed')
|
||||
expect(result.structured).toEqual({ answer: 42 })
|
||||
// Run-scoped runtime: the settle released the last acquisition.
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('a backend unload mid-structured-run settles the run and releases the runtime', async () => {
|
||||
// Rebuild the stack by hand so we hold the backend's fiber.
|
||||
const ctx = new Context()
|
||||
const adapter = new MockAdapter(['hang'])
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await ctx.plugin(AgentRegistry)
|
||||
await ctx.plugin(Invariants)
|
||||
await ctx.plugin(AgentLoop, { agents: [] })
|
||||
await ctx.plugin(SubagentService)
|
||||
const fiber = await ctx.plugin(spawn, { providerName: 'spawn' })
|
||||
ctx.llm.registerAdapter(['mock'], adapter)
|
||||
const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' })
|
||||
const run = ctx.subagents.start('spawn', {
|
||||
prompt: [{ type: 'text', text: 'q' }],
|
||||
parent,
|
||||
outputSchema: { type: 'object', properties: { a: { type: 'number' } } },
|
||||
})
|
||||
// Let the child's step start streaming, then unload the backend. The
|
||||
// backend owns the child agent, so the unload tears the child down and
|
||||
// the run settles — releasing its own runtime acquisition on the way out.
|
||||
await new Promise(resolve => setTimeout(resolve, 30))
|
||||
await fiber.dispose()
|
||||
const result = await run.result
|
||||
expect(result.stopReason).toBe('error')
|
||||
expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
|
||||
await run.dispose()
|
||||
})
|
||||
|
||||
it('has the namespace-plugin export shape (no stray default)', () => {
|
||||
expect('default' in spawn).toBe(false)
|
||||
expect(spawn.name).toBe('subagent-spawn')
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
|
||||
import type { Agent, AgentId, AgentOptions } from '@deepseek-ai/dsh-agent'
|
||||
import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
||||
import type { SchemaSpec } from '@deepseek-ai/dsh-tools'
|
||||
import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
|
||||
|
||||
/**
|
||||
* Which START-TIME features a provider supports. Checked by the service
|
||||
@@ -56,12 +56,16 @@ export interface SubagentStartRequest {
|
||||
/** Per-child agent options (model, system prompt). */
|
||||
agentOptions?: AgentOptions
|
||||
/**
|
||||
* Optional structured-output schema. When set AND the provider's
|
||||
* {@link SubagentCapabilities.outputSchema} is `true`, the child's final
|
||||
* answer is shaped to this schema and surfaced as {@link SubagentResult.structured}.
|
||||
* Optional structured-output schema — an object-rooted JSON Schema within the
|
||||
* enforced subset (see `assertSupportedOutputSchema` in dsh-tools; a schema
|
||||
* outside the subset is rejected loud at start). When set AND the provider's
|
||||
* {@link SubagentCapabilities.outputSchema} is `true`, the child is driven to
|
||||
* report a value matching this schema, surfaced as
|
||||
* {@link SubagentResult.structured}. The schema must be plain host-realm JSON
|
||||
* data — a caller holding foreign-realm data materializes it first.
|
||||
* Requesting it against a provider that lacks the capability is rejected at start.
|
||||
*/
|
||||
outputSchema?: SchemaSpec
|
||||
outputSchema?: StructuredOutputSchema
|
||||
/**
|
||||
* Optional recursion cap (max delegation depth below this child). Requires
|
||||
* {@link SubagentCapabilities.depthLimit}; rejected at start otherwise.
|
||||
@@ -93,6 +97,7 @@ export interface SubagentStopReasonMap {
|
||||
refusal: 'refusal'
|
||||
}
|
||||
|
||||
/** The union over {@link SubagentStopReasonMap} — widens automatically as backends merge in variants. */
|
||||
export type SubagentStopReason = SubagentStopReasonMap[keyof SubagentStopReasonMap]
|
||||
|
||||
/**
|
||||
|
||||
@@ -178,7 +178,7 @@ describe('SubagentService', () => {
|
||||
|
||||
describe('start-time capability validation (fail loud, before any child)', () => {
|
||||
it.each([
|
||||
{ field: 'outputSchema', request: baseRequest({ outputSchema: { x: { type: 'string' } } }) },
|
||||
{ field: 'outputSchema', request: baseRequest({ outputSchema: { type: 'object', properties: { x: { type: 'string' } } } }) },
|
||||
{ field: 'maxDepth', request: baseRequest({ maxDepth: 2 }) },
|
||||
{ field: 'toolFilter', request: baseRequest({ toolFilter: { deny: ['bash'] } }) },
|
||||
])('rejects $field against a provider that lacks the capability — before start() runs', ({ request }) => {
|
||||
@@ -203,7 +203,7 @@ describe('SubagentService', () => {
|
||||
await ctx.plugin(SubagentService)
|
||||
const provider = new StubProvider('strong', ALL_CAPS)
|
||||
ctx.subagents.registerProvider(provider)
|
||||
ctx.subagents.start('strong', baseRequest({ outputSchema: { x: { type: 'string' } }, maxDepth: 1 }))
|
||||
ctx.subagents.start('strong', baseRequest({ outputSchema: { type: 'object', properties: { x: { type: 'string' } } }, maxDepth: 1 }))
|
||||
expect(provider.startCount).toBe(1)
|
||||
})
|
||||
})
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user