fix(pty): close final review gaps
This commit is contained in:
@@ -734,7 +734,7 @@ export const SERVICE_API: readonly ServiceApiEntry[] = [
|
||||
methods: [
|
||||
{
|
||||
signature: 'register(definition: ToolDefinition): () => void',
|
||||
jsDoc: '/**\n * Register globally or in the calling agent scope. Scoped tools shadow\n * globals; duplicates within one layer and the reserved `run_code` name fail.\n * @param definition - the tool schema, execution, and optional presentation functions.\n * @returns the exact disposer that unregisters the tool.\n */',
|
||||
jsDoc: '/**\n * Register globally or in the calling agent scope. Scoped tools shadow\n * globals; duplicates within one layer and the reserved `run_code` name fail.\n * @param definition - tool schema, execution, and optional finalization/presentation callbacks.\n * @returns the exact disposer that unregisters the tool.\n */',
|
||||
},
|
||||
{
|
||||
signature: 'restrict(filter: ToolRestriction): () => void',
|
||||
@@ -758,7 +758,7 @@ export const SERVICE_API: readonly ServiceApiEntry[] = [
|
||||
},
|
||||
{
|
||||
signature: 'async execute(exec: ToolExecutionInput): Promise<ToolExecutionResult>',
|
||||
jsDoc: '/**\n * Execute through pre-policy, guards, around-dispatch, post-policy, and final\n * notification. Tool and listener failures resolve as materialized error\n * results; an invisible tool reports `UNKNOWN_TOOL`. The returned outcome is\n * the same lossless, frozen snapshot final observers receive. Cancellation\n * arriving after entry and before final result materialization skips a\n * not-yet-started body with `ABORTED_BEFORE_DISPATCH` or replaces a\n * successful started outcome with `ABORTED`; already-started work is still\n * drained and may retain a tool-owned structured error.\n * @param exec - the typed same-process call input. The registry assigns its\n * correlation token before policy begins.\n * @returns the materialized final result.\n */',
|
||||
jsDoc: '/**\n * Execute through pre-policy, guards, around-dispatch, post-policy,\n * definition-owned content finalization, and final notification. Tool and\n * listener failures resolve as materialized error results; an invisible tool\n * reports `UNKNOWN_TOOL`. The returned outcome is the same lossless, frozen\n * snapshot final observers receive. Cancellation\n * arriving after entry and before final result materialization skips a\n * not-yet-started body with `ABORTED_BEFORE_DISPATCH` or replaces a\n * successful started outcome with `ABORTED`; already-started work is still\n * drained and may retain a tool-owned structured error.\n * @param exec - the typed same-process call input. The registry assigns its\n * correlation token before policy begins.\n * @returns the materialized final result.\n */',
|
||||
},
|
||||
],
|
||||
},
|
||||
@@ -1973,7 +1973,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [
|
||||
},
|
||||
{
|
||||
name: 'ToolDefinition',
|
||||
declaration: 'export interface ToolDefinition extends ToolSchema {\n execute(args: unknown, exec: ToolRunContext): Promise<ToolExecuteReturn>;\n timeoutMs?: number;\n isConcurrencySafe?(args: unknown): boolean;\n presentCall?(args: unknown): ToolCallView | undefined;\n presentResult?(args: unknown, result: ToolResult): ToolResultView | undefined;\n}',
|
||||
declaration: 'export interface ToolDefinition extends ToolSchema {\n execute(args: unknown, exec: ToolRunContext): Promise<ToolExecuteReturn>;\n finalizeContent?(exec: Readonly<ToolExecution>, result: Readonly<ToolExecutionResult>): ContentBlock[] | undefined;\n timeoutMs?: number;\n isConcurrencySafe?(args: unknown): boolean;\n presentCall?(args: unknown): ToolCallView | undefined;\n presentResult?(args: unknown, result: ToolResult): ToolResultView | undefined;\n}',
|
||||
},
|
||||
{
|
||||
name: 'ToolErrorInfo',
|
||||
|
||||
@@ -76,7 +76,7 @@ describe('cordis_inspect', () => {
|
||||
expect(report).toContain('- tools — Tool registry and execution pipeline.')
|
||||
expect(report).toContain('/**')
|
||||
expect(report).toContain('Register globally or in the calling agent scope.')
|
||||
expect(report).toContain('@param definition - the tool schema')
|
||||
expect(report).toContain('@param definition - tool schema, execution, and optional finalization/presentation callbacks')
|
||||
expect(report).toContain('@returns the exact disposer')
|
||||
expect(report).toContain('register(definition: ToolDefinition)')
|
||||
expect(report).toContain('type shapes (referenced by the signatures above')
|
||||
|
||||
@@ -67,7 +67,7 @@ Within a step, exclusive calls form barriers; parallel-safe calls use a bounded
|
||||
### What belongs to plugins
|
||||
|
||||
Everything that goes beyond "call the model, run the tools, repeat" belongs to plugins listening on the event taxonomy:
|
||||
- Hooks and policy: the relevant `agent/*` checkpoints plus the guarded `tools/pre-execute` → `tools/execute` → `tools/post-execute` → `tools/result` pipeline; exact signatures and modes live in the [generated event catalog](../../../docs/cordis-catalog/events.md)
|
||||
- Hooks and policy: the relevant `agent/*` checkpoints plus the guarded `tools/pre-execute` → `tools/execute` → `tools/post-execute` → definition-owned `finalizeContent` → `tools/result` pipeline; exact event signatures and modes live in the [generated event catalog](../../../docs/cordis-catalog/events.md)
|
||||
- Compaction: pressure on `agent/post-step`; canonical context overflow on `agent/request-error`
|
||||
- Transient model recovery: `dsh-llm-retry` on `agent/request-error`, with finite code-specific budgets and non-surface `llm/retry` status events
|
||||
- Sandbox, permission, plan mode: `tools/pre-execute` for extensible deny/ask, `tools.guard()` for monotonic owner policy, `tools/post-execute` for result decisions, and `tools/result` for final observation
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# dsh-tools
|
||||
|
||||
Tool registry and execution pipeline. Tool plugins register their schemas and executors; the agent loop executes each call through `tools/pre-execute` (the extensible allow/deny gate) → monotonic registered guards → `tools/execute` (an around-dispatch wrapper for timeout/retry/metrics plugins) → `tools/post-execute` (inspect/replace the result, attach context) → the observe-only `tools/result` notification. The registry also owns HOW its tools are presented to the model — its `mode` config selects native function calling, [Code Mode](#code-mode), or both.
|
||||
Tool registry and execution pipeline. Tool plugins register their schemas and executors; the agent loop executes each call through `tools/pre-execute` (the extensible allow/deny gate) → monotonic registered guards → `tools/execute` (an around-dispatch wrapper for timeout/retry/metrics plugins) → `tools/post-execute` (inspect/replace the result, attach context) → the definition-owned `finalizeContent` boundary → the observe-only `tools/result` notification. The registry also owns HOW its tools are presented to the model — its `mode` config selects native function calling, [Code Mode](#code-mode), or both.
|
||||
|
||||
## Service: `ToolRegistry` (ctx key: `tools`)
|
||||
|
||||
@@ -15,7 +15,7 @@ tools:
|
||||
|
||||
### Public API
|
||||
|
||||
- `ctx.tools.register(definition: ToolDefinition): () => void` Register a trusted typed same-process definition. The layer is the calling context's scope: a plain plugin context registers globally; an agent's `agent.ctx` registers for that agent alone, shadowing a same-named global tool there. Duplicate names within one layer throw; non-native modes also reject the reserved `run_code` transport name. `timeoutMs`, when present, must be positive and finite. Disposed with the calling fiber.
|
||||
- `ctx.tools.register(definition: ToolDefinition): () => void` Register a trusted typed same-process definition. The layer is the calling context's scope: a plain plugin context registers globally; an agent's `agent.ctx` registers for that agent alone, shadowing a same-named global tool there. Duplicate names within one layer throw; non-native modes also reject the reserved `run_code` transport name. `timeoutMs`, when present, must be positive and finite. The optional synchronous `finalizeContent` callback is snapshotted when a call starts and may replace only final model-facing content after every pipeline outcome is normalized. Disposed with the calling fiber.
|
||||
- `ctx.tools.restrict(filter)` applies an agent-scoped allow/deny mask to global tools and throws from a plain context. The filter is snapshotted at registration; multiple masks intersect and scope-local tools merge afterwards. Deny masks admit later unnamed globals, while allow masks exclude later names. Unknown, local, or reserved names and empty filters reject. This is live visibility composition, not an authority boundary; see the [scope security non-goal](../../../.agents/notes/implemented/architecture/2026-07-08-agent-scope-contexts.md#security-and-authority-are-explicit-non-goals).
|
||||
- `ctx.tools.get(name: string, scope?: ScopeKey): ToolDefinition | undefined` Resolution as one scope sees it (shadowing applied; a restricted-away global reads as absent) — presenters pass the calling agent so the card matches what executed.
|
||||
- `ctx.tools.schemas(scope?: ScopeKey): ToolSchema[]` Schemas of everything the scope can see (without the `execute` functions). The shipped tools' schemas are catalogued in [docs/tool-catalog.md](../../../docs/tool-catalog.md), generated by booting each tool plugin and harvesting this method (see [the tool-schema-catalog Agent Note](../../../.agents/notes/implemented/process/2026-07-02-tool-schema-catalog.md)).
|
||||
@@ -33,11 +33,11 @@ Cancellation is cooperative and quiescent. Every typed invocation supplies a cal
|
||||
|
||||
### Live events
|
||||
|
||||
The live registry pipeline has three transformable waterfalls followed by the observe-only `tools/result` boundary; registry changes are deliberately unfiltered shared-state notifications. Exact signatures, dispatch modes, scope filtering, and failure-containment contracts live in the generated [Cordis event catalog](../../../docs/cordis-catalog/events.md), while the complete ordering is visualized in the generated [tool execution pipeline](../../../docs/tool-execution-pipeline.md). `tools/result` is live; the similarly named `tool/result` is the durable session event the agent loop appends afterwards.
|
||||
The live registry pipeline has three transformable waterfalls, then the definition-owned content finalizer, then the observe-only `tools/result` boundary; registry changes are deliberately unfiltered shared-state notifications. Exact signatures, dispatch modes, scope filtering, and failure-containment contracts live in the generated [Cordis event catalog](../../../docs/cordis-catalog/events.md), while the complete ordering is visualized in the generated [tool execution pipeline](../../../docs/tool-execution-pipeline.md). `tools/result` is live; the similarly named `tool/result` is the durable session event the agent loop appends afterwards.
|
||||
|
||||
### Key types
|
||||
|
||||
- `ToolDefinition` — `ToolSchema` + `execute(args, exec)`, whose async work must cooperatively stop through `exec.signal`, plus optional presentation callbacks, cooperative `timeoutMs`, and optional per-call `isConcurrencySafe(args)` classification.
|
||||
- `ToolDefinition` — `ToolSchema` + `execute(args, exec)`, whose async work must cooperatively stop through `exec.signal`, plus optional final-content and presentation callbacks, cooperative `timeoutMs`, and optional per-call `isConcurrencySafe(args)` classification. `finalizeContent(exec, result)` runs exactly once for every normalized result, including failures that bypass post-policy, and can replace only `content`; it must be synchronous and total.
|
||||
- `ToolExecutionInput` — the caller-supplied call description: `{ callId, name, arguments, signal, agent?, parent? }`; `signal` is required and readonly, callers may pass an enclosing execution's opaque token as `parent`, and callers never choose the new execution's own token.
|
||||
- `ToolExecutionToken` — a fresh branded `Symbol` assigned by the registry. It supports equality correlation only and never crosses a model, log, or worker boundary.
|
||||
- `ToolExecution` — the readonly pipeline view: immutable `{ token, callId, name, arguments, signal, agent?, parent? }`; the registry separately retains and re-fuses the original caller signal. `ToolDispatchExecution` is the `tools/execute`-only view whose required signal is mutable, so a wrapper may replace and restore it but cannot delete it. A nested call's `parent` is a `ToolExecutionToken`, not an execution object.
|
||||
@@ -53,7 +53,7 @@ The live registry pipeline has three transformable waterfalls followed by the ob
|
||||
- Tool plugins call `ctx.tools.register()` — schemas flow into the assembly automatically.
|
||||
- `tools/pre-execute` is the reorderable allow/deny/ask gate; `ctx.tools.guard()` adds monotonic owner policy after it.
|
||||
- `tools/execute` wraps normalized core dispatch for timeout, retry, or metrics. Wrappers may replace only the operational signal.
|
||||
- `tools/post-execute` may replace content, block with feedback, or attach ordered contexts; `tools/result` observes the immutable final outcome.
|
||||
- `tools/post-execute` may replace content, block with feedback, or attach ordered contexts. A definition's optional `finalizeContent` then owns its last content-only invariant across normal results and outer pipeline failures; `tools/result` observes the immutable final outcome.
|
||||
- Exact signatures and ordering live in the generated [event catalog](../../../docs/cordis-catalog/events.md) and [pipeline](../../../docs/tool-execution-pipeline.md).
|
||||
- MCP servers: one plugin per server, discover tools, call `ctx.tools.register()` with the server's schemas.
|
||||
|
||||
|
||||
@@ -139,6 +139,18 @@ export interface ToolDefinition extends ToolSchema {
|
||||
* @returns model-facing content plus optional private presentation metadata.
|
||||
*/
|
||||
execute(args: unknown, exec: ToolRunContext): Promise<ToolExecuteReturn>
|
||||
/**
|
||||
* Synchronous last-mile transform for model-facing content. The registry
|
||||
* snapshots this callback when execution starts and invokes it exactly once
|
||||
* for every normalized outcome, including pipeline failures that bypass
|
||||
* `tools/post-execute`, immediately before lossless materialization.
|
||||
* Returning `undefined` preserves the content; every other result field
|
||||
* remains registry-owned. The callback must be total and must not throw.
|
||||
* @param exec - immutable execution identity and arguments.
|
||||
* @param result - complete normalized outcome before materialization.
|
||||
* @returns replacement content, or `undefined` to preserve it.
|
||||
*/
|
||||
finalizeContent?(exec: Readonly<ToolExecution>, result: Readonly<ToolExecutionResult>): ContentBlock[] | undefined
|
||||
/**
|
||||
* Cooperative tool-call timeout budget in milliseconds. Omit for no deadline.
|
||||
* Enforced by `@deepseek-ai/dsh-timeout-policy` (a `tools/execute` wrapper); it
|
||||
@@ -301,9 +313,9 @@ export interface ToolRegistryScheduler {
|
||||
prepare(exec: ToolExecutionInput): Promise<ScheduledToolPreparation>
|
||||
/** Run only the around-dispatch/body stage. */
|
||||
dispatch(exec: ToolRunContext): Promise<ScheduledToolDispatch>
|
||||
/** Run ordered post-execute finalization, then materialize and notify the final outcome. */
|
||||
/** Run post-execute and definition-owned content finalization, then materialize and notify. */
|
||||
finalize(exec: ToolRunContext, result: ToolExecutionResult): Promise<ToolExecutionResult>
|
||||
/** Materialize and notify a final outcome that must bypass post-execute. */
|
||||
/** Run definition-owned content finalization, then materialize and notify without post-execute. */
|
||||
finish(exec: ToolRunContext, result: ToolExecutionResult): ToolExecutionResult
|
||||
}
|
||||
|
||||
@@ -540,6 +552,8 @@ export class ToolRegistry extends Service {
|
||||
private deferredContexts = new WeakMap<ToolRunContext, HookContext[]>()
|
||||
/** Original caller cancellation, kept outside the wrapper-mutable execution object. */
|
||||
private cancellationStates = new WeakMap<ToolRunContext, ToolCancellationState>()
|
||||
/** Definition-owned final content transform snapshotted before policy begins. */
|
||||
private contentFinalizers = new WeakMap<ToolRunContext, ToolDefinition['finalizeContent']>()
|
||||
private readonly layers = new ScopedLayers(
|
||||
scope => new ToolLayer(scope),
|
||||
() => { this.ctx.emit('tools/change') },
|
||||
@@ -617,7 +631,7 @@ export class ToolRegistry extends Service {
|
||||
/**
|
||||
* Register globally or in the calling agent scope. Scoped tools shadow
|
||||
* globals; duplicates within one layer and the reserved `run_code` name fail.
|
||||
* @param definition - the tool schema, execution, and optional presentation functions.
|
||||
* @param definition - tool schema, execution, and optional finalization/presentation callbacks.
|
||||
* @returns the exact disposer that unregisters the tool.
|
||||
*/
|
||||
register(definition: ToolDefinition): () => void {
|
||||
@@ -784,10 +798,11 @@ export class ToolRegistry extends Service {
|
||||
}
|
||||
|
||||
/**
|
||||
* Execute through pre-policy, guards, around-dispatch, post-policy, and final
|
||||
* notification. Tool and listener failures resolve as materialized error
|
||||
* results; an invisible tool reports `UNKNOWN_TOOL`. The returned outcome is
|
||||
* the same lossless, frozen snapshot final observers receive. Cancellation
|
||||
* Execute through pre-policy, guards, around-dispatch, post-policy,
|
||||
* definition-owned content finalization, and final notification. Tool and
|
||||
* listener failures resolve as materialized error results; an invisible tool
|
||||
* reports `UNKNOWN_TOOL`. The returned outcome is the same lossless, frozen
|
||||
* snapshot final observers receive. Cancellation
|
||||
* arriving after entry and before final result materialization skips a
|
||||
* not-yet-started body with `ABORTED_BEFORE_DISPATCH` or replaces a
|
||||
* successful started outcome with `ABORTED`; already-started work is still
|
||||
@@ -826,6 +841,8 @@ export class ToolRegistry extends Service {
|
||||
const agent = exec.agent
|
||||
const parent = exec.parent
|
||||
const signal = exec.signal
|
||||
const definition = this.get(name, agent)
|
||||
const finalizeContent = definition?.finalizeContent?.bind(definition)
|
||||
const base = {
|
||||
token,
|
||||
callId,
|
||||
@@ -844,6 +861,7 @@ export class ToolRegistry extends Service {
|
||||
}
|
||||
const execution: MutableToolRunContext = { ...base, arguments: deepFreeze(detached) }
|
||||
this.deferredContexts.set(execution, deferredContexts)
|
||||
this.contentFinalizers.set(execution, finalizeContent)
|
||||
this.cancellationStates.set(execution, {
|
||||
callerSignal: signal,
|
||||
bodyInvoked: false,
|
||||
@@ -851,6 +869,7 @@ export class ToolRegistry extends Service {
|
||||
return { kind: 'ready', exec: execution }
|
||||
} catch (error: unknown) {
|
||||
const execution: MutableToolRunContext = { ...base, arguments: undefined }
|
||||
this.contentFinalizers.set(execution, finalizeContent)
|
||||
return { kind: 'final-result', exec: execution, result: toolErrorResult(error) }
|
||||
}
|
||||
}
|
||||
@@ -1008,7 +1027,8 @@ export class ToolRegistry extends Service {
|
||||
}
|
||||
|
||||
/**
|
||||
* Run ordered post-execute, then materialize and notify the final outcome.
|
||||
* Run ordered post-execute, then apply definition-owned content finalization,
|
||||
* materialize, and notify the final outcome.
|
||||
* @param exec - the prepared execution.
|
||||
* @param result - dispatch/pre result that still needs post-execute.
|
||||
* @returns the materialized final result.
|
||||
@@ -1029,7 +1049,8 @@ export class ToolRegistry extends Service {
|
||||
}
|
||||
|
||||
/**
|
||||
* Materialize and notify a final result that must bypass post-execute.
|
||||
* Apply definition-owned content finalization, then materialize and notify a
|
||||
* final result that must bypass post-execute.
|
||||
* @param exec - the prepared execution.
|
||||
* @param result - final result.
|
||||
* @returns the materialized final result.
|
||||
@@ -1038,7 +1059,7 @@ export class ToolRegistry extends Service {
|
||||
private finishScheduledExecution(exec: ToolRunContext, result: ToolExecutionResult): ToolExecutionResult {
|
||||
let finalResult: ToolExecutionResult
|
||||
try {
|
||||
finalResult = this.materializeFinalResult(result)
|
||||
finalResult = this.materializeFinalResult(this.applyFinalContent(exec, result))
|
||||
} catch (error: unknown) {
|
||||
finalResult = this.materializeFinalResult(toolErrorResult(error))
|
||||
}
|
||||
@@ -1046,6 +1067,14 @@ export class ToolRegistry extends Service {
|
||||
return finalResult
|
||||
}
|
||||
|
||||
/** Apply the snapshotted tool-owned content transform without exposing other result fields. */
|
||||
private applyFinalContent(exec: ToolRunContext, result: ToolExecutionResult): ToolExecutionResult {
|
||||
const finalizeContent = this.contentFinalizers.get(exec)
|
||||
if (finalizeContent === undefined) return result
|
||||
const content = finalizeContent(exec, result)
|
||||
return content === undefined ? result : { ...result, content }
|
||||
}
|
||||
|
||||
/** Notify observers without exposing a mutation or error channel into the outcome. */
|
||||
private notifyResult(exec: ToolExecution, result: ToolExecutionResult): void {
|
||||
// Freeze the registry's live object before observers receive its readonly
|
||||
|
||||
@@ -1,7 +1,15 @@
|
||||
/** Typed tool-parameter DSL with argument inference and JSON Schema output. @module dsh-tools/schema */
|
||||
|
||||
import { assertNever, HarnessError } from '@deepseek-ai/dsh-llm'
|
||||
import type { ToolDefinition, ToolExecuteReturn, ToolRunContext, ToolResult } from './index.ts'
|
||||
import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
||||
import type {
|
||||
ToolDefinition,
|
||||
ToolExecuteReturn,
|
||||
ToolExecution,
|
||||
ToolExecutionResult,
|
||||
ToolRunContext,
|
||||
ToolResult,
|
||||
} from './index.ts'
|
||||
import type { ToolCallView, ToolResultView } from './presentation.ts'
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -294,6 +302,15 @@ export interface DefineToolOptions<S extends SchemaSpec> {
|
||||
* presentation payload (see {@link ToolExecuteReturn}).
|
||||
*/
|
||||
execute(args: InferArgs<S>, exec: ToolRunContext): Promise<ToolExecuteReturn>
|
||||
/**
|
||||
* Optional last-mile content transform for every normalized outcome. Unlike
|
||||
* `execute`, arguments remain `unknown` because invalid-input failures also
|
||||
* reach this callback. See {@link ToolDefinition.finalizeContent}.
|
||||
* @param exec - immutable execution identity and arguments.
|
||||
* @param result - complete normalized outcome before materialization.
|
||||
* @returns replacement content, or `undefined` to preserve it.
|
||||
*/
|
||||
finalizeContent?(exec: Readonly<ToolExecution>, result: Readonly<ToolExecutionResult>): ContentBlock[] | undefined
|
||||
/**
|
||||
* Optional: how to present the PENDING state of one call in a UI (an editor
|
||||
* tool-call card, a CLI log line). `args` is the typed, schema-validated
|
||||
@@ -317,7 +334,7 @@ export interface DefineToolOptions<S extends SchemaSpec> {
|
||||
* inferred from its per-property schema. Raw JSON-Schema definitions remain
|
||||
* valid inputs to {@link ToolRegistry.register}; this helper is authoring sugar.
|
||||
* @param options - the tool's name, description, typed parameter schema,
|
||||
* execute body, and optional presenters.
|
||||
* execute body, and optional finalization/presentation callbacks.
|
||||
* @returns a registry-ready definition with strict execution validation and
|
||||
* soft presenter and classifier validation for replay compatibility.
|
||||
*/
|
||||
@@ -326,6 +343,8 @@ export function defineTool<S extends SchemaSpec>(options: DefineToolOptions<S>):
|
||||
// eslint-disable-next-line @typescript-eslint/unbound-method
|
||||
const userExecute = options.execute
|
||||
// eslint-disable-next-line @typescript-eslint/unbound-method
|
||||
const userFinalizeContent = options.finalizeContent
|
||||
// eslint-disable-next-line @typescript-eslint/unbound-method
|
||||
const userPresentCall = options.presentCall
|
||||
// eslint-disable-next-line @typescript-eslint/unbound-method
|
||||
const userPresentResult = options.presentResult
|
||||
@@ -349,6 +368,9 @@ export function defineTool<S extends SchemaSpec>(options: DefineToolOptions<S>):
|
||||
return userExecute(args as InferArgs<S>, exec)
|
||||
},
|
||||
}
|
||||
if (userFinalizeContent) {
|
||||
tool.finalizeContent = (exec, result) => userFinalizeContent(exec, result)
|
||||
}
|
||||
// Presentation is display-only and may run on REPLAY of arbitrary logged args
|
||||
// (possibly from an older schema), so it must never throw: validate softly and
|
||||
// fall back to `undefined` (a generic UI presentation) on any mismatch, rather
|
||||
|
||||
@@ -47,22 +47,24 @@ describe('ToolRegistry', () => {
|
||||
expect(assembly.tools.map(t => t.name)).toEqual(['echo'])
|
||||
})
|
||||
|
||||
it('schemas() drops the UI presentation callbacks — they must never reach the model', async () => {
|
||||
it('schemas() drops host callbacks — they must never reach the model', async () => {
|
||||
const ctx = await setup()
|
||||
// A tool that declares presentCall/presentResult (functions). schemas() feeds
|
||||
// the system-prompt assembly → the model request, so those callbacks (and
|
||||
// `execute`) must be stripped: a function in the JSON tool schema would
|
||||
// corrupt the request. schemas() is an explicit allowlist, so it can't leak.
|
||||
// A tool that declares finalization and presentation functions. schemas()
|
||||
// feeds the system-prompt assembly → the model request, so every callback
|
||||
// (including `execute`) must be stripped: a function in the JSON tool schema
|
||||
// would corrupt the request. schemas() is an explicit allowlist, so it can't leak.
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'present',
|
||||
description: 'has presenters',
|
||||
parameters: { x: { type: 'string', required: true } },
|
||||
async execute() { return [] },
|
||||
finalizeContent: (_exec, result) => result.content,
|
||||
presentCall: args => ({ card: 'generic', title: args.x }),
|
||||
presentResult: (args, result) => ({ card: 'generic', title: args.x, content: result.content }),
|
||||
}))
|
||||
const schema = ctx.tools.schemas()[0] as unknown as Record<string, unknown>
|
||||
expect(Object.keys(schema).sort()).toEqual(['description', 'name', 'parameters'])
|
||||
expect(schema.finalizeContent).toBeUndefined()
|
||||
expect(schema.presentCall).toBeUndefined()
|
||||
expect(schema.presentResult).toBeUndefined()
|
||||
expect(schema.execute).toBeUndefined()
|
||||
@@ -388,6 +390,33 @@ describe('ToolRegistry', () => {
|
||||
expect(result.content[0]).toMatchObject({ text: 'output rejected: try again' })
|
||||
})
|
||||
|
||||
it('runs the snapshotted final content transform after outer pipeline normalization', async () => {
|
||||
const ctx = await setup()
|
||||
const dispose = ctx.tools.register(defineTool({
|
||||
name: 'bounded',
|
||||
description: 'bounded result',
|
||||
parameters: {},
|
||||
async execute() { return [{ type: 'text', text: 'body' }] },
|
||||
finalizeContent(exec, result) {
|
||||
expect(exec.name).toBe('bounded')
|
||||
expect(result.isError).toBe(true)
|
||||
return [{ type: 'text', text: 'bounded failure' }]
|
||||
},
|
||||
}))
|
||||
ctx.on('tools/pre-execute', async () => {
|
||||
dispose()
|
||||
throw new HarnessError('policy failed', 'POLICY_FAILED')
|
||||
})
|
||||
|
||||
const result = await ctx.tools.execute({ signal: testToolSignal, callId: CallId('bounded'), name: 'bounded', arguments: {} })
|
||||
|
||||
expect(result).toEqual({
|
||||
content: [{ type: 'text', text: 'bounded failure' }],
|
||||
isError: true,
|
||||
error: { name: 'HarnessError', code: 'POLICY_FAILED' },
|
||||
})
|
||||
})
|
||||
|
||||
it('a block decision can ALSO attach additionalContexts', async () => {
|
||||
const ctx = await setup()
|
||||
ctx.tools.register(echoTool)
|
||||
|
||||
@@ -7,7 +7,7 @@ Owner-scoped persistent PTY seam. `PtyService` registers as `ctx.pty`, mints opa
|
||||
- Backends register one stable `type` and return an unpublished `PtyBackendSession`; failed or cancelled setup must clean partial resources, and a failed cleanup rejects with `PtyBackendCleanupError` so the registry can retain it across cancellation.
|
||||
- Spawn cancellation preserves the caller's exact abort reason. Service disposal and owner loss remain distinct machine-routable failures after backend setup.
|
||||
- Owner and service disposal abort unpublished setup through a service-owned signal and await backend settlement plus rollback before returning.
|
||||
- A service rollback or backend-reported startup cleanup failure rejects the disposing lifecycle instead of claiming quiescence; the spawn caller still receives its exact cancellation reason.
|
||||
- A rollback-close or backend-reported startup cleanup failure rejects the disposing lifecycle instead of claiming quiescence. Caller-triggered cancellation still receives its exact reason; lifecycle-triggered rollback failure also rejects the pending spawn.
|
||||
- A backend cleanup failure that follows caller cancellation remains owner activity until owner or service disposal consumes and reports it, so lifecycle policy cannot mistake failed cleanup for quiescence.
|
||||
- `hasOwnerActivity(owner)` spans unpublished setup through final close, so lifecycle policy can fence the exact owner without a publication race.
|
||||
- A successful spawn publishes one `PtySessionId`. The optional `name` is owner-local display metadata, never authority.
|
||||
|
||||
@@ -213,7 +213,7 @@ export class PtyService extends Service {
|
||||
} catch (cancellation: unknown) {
|
||||
failure = cancellation
|
||||
}
|
||||
if (rollbackFailure !== undefined) {
|
||||
if (rollbackFailure !== undefined && signal?.aborted !== true) {
|
||||
throw new AggregateError([failure, rollbackFailure.error], 'PTY spawn and rollback both failed')
|
||||
}
|
||||
throw failure
|
||||
|
||||
@@ -233,6 +233,29 @@ describe('PtyService ownership and lifecycle', () => {
|
||||
expect(ctx.agents.get(owner.id)).toBe(owner)
|
||||
})
|
||||
|
||||
it('preserves caller cancellation when unpublished rollback fails', async () => {
|
||||
const ctx = await harness()
|
||||
const gate = Promise.withResolvers<PtyBackendSession>()
|
||||
const session = new StubSession()
|
||||
session.rejectClose = true
|
||||
ctx.pty.registerBackend({ type: 'slow', spawn: () => gate.promise })
|
||||
const owner = stubAgent(ctx, 'owner')
|
||||
ctx.agents.register(owner)
|
||||
const controller = new AbortController()
|
||||
const reason = new Error('cancelled by caller')
|
||||
|
||||
const pending = ctx.pty.spawn(owner, { type: 'slow' }, controller.signal)
|
||||
controller.abort(reason)
|
||||
gate.resolve(session)
|
||||
|
||||
await expect(pending).rejects.toBe(reason)
|
||||
expect(ctx.pty.hasOwnerActivity(owner)).toBe(true)
|
||||
const internal = ctx.pty as unknown as { disposeAll(): Promise<void> }
|
||||
await expect(internal.disposeAll()).rejects.toThrow('failed to clean up PTY lifecycle')
|
||||
expect(ctx.pty.hasOwnerActivity(owner)).toBe(false)
|
||||
expect(session.closed).toEqual(['PTY spawn rolled back'])
|
||||
})
|
||||
|
||||
it('preserves caller cancellation when a backend rejects in response to it', async () => {
|
||||
const ctx = await harness()
|
||||
const started = Promise.withResolvers<undefined>()
|
||||
|
||||
@@ -11,7 +11,7 @@ Six model-facing tools over `ctx.pty`: `terminal_open`, `terminal_send`, `termin
|
||||
| `enableRunInBackground` | `true` | expose and accept `run_in_background`; false omits the schema field and rejects a forced undeclared argument |
|
||||
| `maxResultBytes` | `262144` | UTF-8 cap (minimum `64`) for each complete terminal result or PTY task output after wait, session, pagination, truncation, and task-status metadata |
|
||||
|
||||
Both values are validated at load. The minimum result cap keeps every registry-issued session or task id visible in its creation acknowledgement. When a result exceeds `maxResultBytes`, rendering reserves space for control metadata and a truncation marker when they fit; cuts preserve UTF-8 boundaries. An outer `tools/post-execute` wrapper applies the same cap after a terminal pre-execute denial or single-text post-execute replacement/block; a structured multi-block policy result retains its shape.
|
||||
Both values are validated at load. The minimum result cap keeps every registry-issued session or task id visible in its creation acknowledgement. When a result exceeds `maxResultBytes`, rendering reserves space for control metadata and a truncation marker when they fit; cuts preserve UTF-8 boundaries. Each terminal definition's final-content callback applies the same cap after normalized pre-, around-, and post-execute policy failures, denials, short-circuits, replacements, or blocks; a structured multi-block policy result retains its shape.
|
||||
|
||||
## Model Experience
|
||||
|
||||
@@ -53,7 +53,7 @@ Prefix-stable while tool visibility and definitions are unchanged.
|
||||
|
||||
#### What the model sees
|
||||
|
||||
Spawn returns the id and bounded MOTD. Send/read return bounded terminal text plus readiness/history markers. Background mode returns a generic task id. Every terminal-owned or policy-produced single-text result is capped by `maxResultBytes` after normalized errors, denials, replacements, blocks, and generic task status text. Structured multi-block policy results retain their shape. Results remain in session history until compaction; incremental task reads do not repeat consumed output.
|
||||
Spawn returns the id and bounded MOTD. Send/read return bounded terminal text plus readiness/history markers. Background mode returns a generic task id. Every terminal-owned or policy-produced single-text result is capped by `maxResultBytes` after normalized tool or pipeline errors, denials, short-circuits, replacements, blocks, and generic task status text. Structured multi-block policy results retain their shape. Results remain in session history until compaction; incremental task reads do not repeat consumed output.
|
||||
|
||||
#### Token effect
|
||||
|
||||
|
||||
@@ -12,7 +12,7 @@ import { PtySessionId } from '@deepseek-ai/dsh-pty'
|
||||
import type { PtySendResult, PtySessionId as PtySessionIdType, PtySignal } from '@deepseek-ai/dsh-pty'
|
||||
import type {} from '@deepseek-ai/dsh-tasks'
|
||||
import { defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import type { PostToolDecision, ToolExecutionResult } from '@deepseek-ai/dsh-tools'
|
||||
import type { ToolDefinition, ToolExecutionResult } from '@deepseek-ai/dsh-tools'
|
||||
import { boundTerminalText, renderList, renderRead, renderSend, renderSendRead, renderSpawn } from './render.ts'
|
||||
|
||||
declare module '@deepseek-ai/dsh-tasks' {
|
||||
@@ -31,15 +31,6 @@ export const DEFAULT_MAX_RESULT_BYTES = 256 * 1024
|
||||
/** Smallest cap that preserves every counter-backed PTY and task id in its creation acknowledgement. */
|
||||
export const MIN_MAX_RESULT_BYTES = 64
|
||||
|
||||
const TOOL_NAMES = new Set([
|
||||
'terminal_open',
|
||||
'terminal_send',
|
||||
'terminal_read',
|
||||
'terminal_signal',
|
||||
'terminal_close',
|
||||
'terminal_list',
|
||||
])
|
||||
|
||||
/** Model-facing terminal tool configuration. */
|
||||
export interface Config {
|
||||
/** Expose `run_in_background` and accept background sends (default true). */
|
||||
@@ -114,17 +105,10 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
if (!Number.isSafeInteger(maxResultBytes) || maxResultBytes < MIN_MAX_RESULT_BYTES) {
|
||||
throw new Error(`tool-pty: maxResultBytes must be a safe integer of at least ${MIN_MAX_RESULT_BYTES}`)
|
||||
}
|
||||
ctx.on('tools/post-execute', async (exec, result, next): Promise<PostToolDecision> => {
|
||||
const decision = await next()
|
||||
if (!TOOL_NAMES.has(exec.name)) return decision
|
||||
const content = decision.kind === 'block' ? decision.feedback : decision.content ?? result.content
|
||||
const raw = rawContentText(content)
|
||||
if (raw === undefined) return decision
|
||||
const bounded = textResult(raw, maxResultBytes)
|
||||
return decision.kind === 'block'
|
||||
? { ...decision, feedback: bounded }
|
||||
: { ...decision, content: bounded }
|
||||
}, { prepend: true })
|
||||
const finalizeContent: NonNullable<ToolDefinition['finalizeContent']> = (_exec, result) => {
|
||||
const raw = rawContentText(result.content)
|
||||
return raw === undefined ? undefined : textResult(raw, maxResultBytes)
|
||||
}
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tool:pty',
|
||||
order: 106,
|
||||
@@ -139,6 +123,7 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
name: { type: 'string', description: 'Optional owner-local display name such as "main" or "gdb".' },
|
||||
cwd: { type: 'string', description: 'Initial working directory. Defaults to the deployment workspace root.' },
|
||||
},
|
||||
finalizeContent,
|
||||
async execute(args: SpawnArgs, exec) {
|
||||
if (args.type.length === 0) throw new Error('type must be a non-empty string')
|
||||
const result = await ctx.pty.spawn(requireAgent(exec.agent), {
|
||||
@@ -166,6 +151,7 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
? { run_in_background: { type: 'boolean' as const, description: 'Return a task id immediately; collect with task_output or stop with task_kill.' } }
|
||||
: {},
|
||||
},
|
||||
finalizeContent,
|
||||
async execute(args: SendArgs, exec): Promise<ToolExecutionResult> {
|
||||
const owner = requireAgent(exec.agent)
|
||||
const id = sessionId(args)
|
||||
@@ -224,6 +210,7 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
offset: { type: 'number', description: 'Newest-relative line offset (default 0).' },
|
||||
count: { type: 'number', description: 'Requested line count (default 500; backend caps apply).' },
|
||||
},
|
||||
finalizeContent,
|
||||
execute(args: ReadArgs, exec) {
|
||||
const result = ctx.pty.read(requireAgent(exec.agent), sessionId(args), {
|
||||
...args.offset !== undefined ? { offset: args.offset } : {},
|
||||
@@ -241,6 +228,7 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
sessionId: { type: 'string', required: true, description: 'Terminal session id.' },
|
||||
signal: { type: 'string', required: true, enum: ['SIGINT', 'SIGTERM', 'SIGKILL', 'SIGTSTP', 'SIGHUP'], description: 'Signal to deliver. Shell-targeted SIGKILL is rejected; use terminal_close.' },
|
||||
},
|
||||
finalizeContent,
|
||||
async execute(args: SignalArgs, exec) {
|
||||
const result = await ctx.pty.signal(requireAgent(exec.agent), sessionId(args), args.signal)
|
||||
return textResult(`delivered ${args.signal} to foreground process group ${result.targetPgid}`, maxResultBytes)
|
||||
@@ -254,6 +242,7 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
parameters: {
|
||||
sessionId: { type: 'string', required: true, description: 'Terminal session id.' },
|
||||
},
|
||||
finalizeContent,
|
||||
async execute(args: SessionArgs, exec) {
|
||||
const id = sessionId(args)
|
||||
const closed = await ctx.pty.kill(requireAgent(exec.agent), id)
|
||||
@@ -266,6 +255,7 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
name: 'terminal_list',
|
||||
description: 'List persistent terminal sessions owned by the current agent.',
|
||||
parameters: {},
|
||||
finalizeContent,
|
||||
execute(_args: Record<string, never>, exec) {
|
||||
return Promise.resolve(textResult(renderList(ctx.pty.list(requireAgent(exec.agent)), maxResultBytes), maxResultBytes))
|
||||
},
|
||||
|
||||
@@ -217,11 +217,17 @@ describe('tool-pty foreground surface', () => {
|
||||
expect(Buffer.byteLength(text(background))).toBeLessThanOrEqual(64)
|
||||
})
|
||||
|
||||
it('bounds terminal results after pre- and post-execute policy', async () => {
|
||||
it('bounds terminal results after policy decisions and pipeline failures', async () => {
|
||||
const { ctx, agent } = await setup(false, { maxResultBytes: 64 })
|
||||
ctx.on('tools/pre-execute', async (exec, next) => exec.name === 'terminal_list'
|
||||
? { kind: 'deny', reason: 'd'.repeat(1_000) }
|
||||
: next())
|
||||
ctx.on('tools/pre-execute', async (exec, next) => {
|
||||
if (exec.name === 'terminal_list') return { kind: 'deny', reason: 'd'.repeat(1_000) }
|
||||
if (exec.name === 'terminal_signal') throw new Error(`pre failed: ${'p'.repeat(1_000)}`)
|
||||
return next()
|
||||
})
|
||||
ctx.on('tools/execute', async (exec, next) => {
|
||||
if (exec.name === 'terminal_close') throw new Error(`around failed: ${'e'.repeat(1_000)}`)
|
||||
return next()
|
||||
})
|
||||
ctx.on('tools/post-execute', async (exec, _result, next) => {
|
||||
if (exec.name === 'terminal_open') {
|
||||
return { kind: 'accept', content: [{ type: 'text', text: 'a'.repeat(1_000) }] }
|
||||
@@ -229,6 +235,7 @@ describe('tool-pty foreground surface', () => {
|
||||
if (exec.name === 'terminal_read') {
|
||||
return { kind: 'block', feedback: [{ type: 'text', text: 'b'.repeat(1_000) }] }
|
||||
}
|
||||
if (exec.name === 'terminal_send') throw new Error(`post failed: ${'o'.repeat(1_000)}`)
|
||||
return next()
|
||||
})
|
||||
|
||||
@@ -246,6 +253,17 @@ describe('tool-pty foreground surface', () => {
|
||||
expect(blocked.isError).toBe(true)
|
||||
expect(Buffer.byteLength(text(blocked))).toBeLessThanOrEqual(64)
|
||||
expect(text(blocked)).toContain('[output truncated]')
|
||||
|
||||
const failures = [
|
||||
await call(ctx, 'terminal_signal', { sessionId: 'pty-1', signal: 'SIGINT' }, agent),
|
||||
await call(ctx, 'terminal_close', { sessionId: 'pty-1' }, agent),
|
||||
await call(ctx, 'terminal_send', { sessionId: 'pty-1', text: 'work' }, agent),
|
||||
]
|
||||
for (const failure of failures) {
|
||||
expect(failure.isError).toBe(true)
|
||||
expect(Buffer.byteLength(text(failure))).toBeLessThanOrEqual(64)
|
||||
expect(text(failure)).toContain('[output truncated]')
|
||||
}
|
||||
})
|
||||
|
||||
it('leaves a structured around-dispatch replacement unchanged', async () => {
|
||||
|
||||
@@ -10,7 +10,7 @@ The model-facing control surface for `ctx.tasks`: three kind-independent tools,
|
||||
|
||||
All three use generic ACP cards: `read` for output and list, `execute` for kill.
|
||||
|
||||
When a producer supplies `outputLimitBytes`, `task_output`, terminal `task_kill`, and completion notices cap the complete UTF-8 result after adding status or notice text. Reads retain the output tail and control suffix when they fit; a bounded completion notice instead reserves `background task <id>` and the `task_output` collection instruction before spending remaining bytes on its variable kind, label, status, and detail. An outer pre/post-execute pair captures the caller-visible task before policy and applies its producer cap to single-text denials, around-dispatch short-circuits, normalized task-control failures, replacements, and blocks; structured multi-block policy results retain their shape. An existing producer truncation marker is reused rather than duplicated. Producers that omit the field retain the existing unbounded control-surface behavior.
|
||||
When a producer supplies `outputLimitBytes`, `task_output`, terminal `task_kill`, and completion notices cap the complete UTF-8 result after adding status or notice text. Reads retain the output tail and control suffix when they fit; a bounded completion notice instead reserves `background task <id>` and the `task_output` collection instruction before spending remaining bytes on its variable kind, label, status, and detail. A prepended pre-execute listener captures the caller-visible task before policy, and each task-control definition's final-content callback applies its producer cap to single-text denials, short-circuits, normalized tool or pipeline failures, replacements, and blocks; structured multi-block policy results retain their shape. An existing producer truncation marker is reused rather than duplicated. Producers that omit the field retain the existing unbounded control-surface behavior.
|
||||
|
||||
## Completion notices
|
||||
|
||||
|
||||
@@ -11,7 +11,7 @@ import z from 'schemastery'
|
||||
import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
||||
import { TextRetainer } from '@deepseek-ai/dsh-retention'
|
||||
import { defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import type { GenericCallView, PostToolDecision, ToolExecution } from '@deepseek-ai/dsh-tools'
|
||||
import type { GenericCallView, ToolDefinition, ToolExecution } from '@deepseek-ai/dsh-tools'
|
||||
import { TaskId } from '@deepseek-ai/dsh-tasks'
|
||||
import type { TaskSnapshot } from '@deepseek-ai/dsh-tasks'
|
||||
import type {} from '@deepseek-ai/dsh-system-prompt'
|
||||
@@ -128,18 +128,11 @@ export function apply(ctx: Context, config: Config): void {
|
||||
if (maxBytes !== undefined) outputLimits.set(exec, maxBytes)
|
||||
return next()
|
||||
}, { prepend: true })
|
||||
ctx.on('tools/post-execute', async (exec, result, next): Promise<PostToolDecision> => {
|
||||
const decision = await next()
|
||||
const maxBytes = outputLimits.get(exec)
|
||||
const finalizeTaskContent: NonNullable<ToolDefinition['finalizeContent']> = (exec, result) => {
|
||||
const maxBytes = outputLimits.get(exec) ?? visibleOutputLimit(ctx, exec)
|
||||
outputLimits.delete(exec)
|
||||
if (maxBytes === undefined) return decision
|
||||
const content = decision.kind === 'block' ? decision.feedback : decision.content ?? result.content
|
||||
const bounded = boundSingleText(content, maxBytes)
|
||||
if (bounded === undefined) return decision
|
||||
return decision.kind === 'block'
|
||||
? { ...decision, feedback: bounded }
|
||||
: { ...decision, content: bounded }
|
||||
}, { prepend: true })
|
||||
return maxBytes === undefined ? undefined : boundSingleText(result.content, maxBytes)
|
||||
}
|
||||
|
||||
// Producers may start work only while a control surface is attached.
|
||||
ctx.tasks.attachSurface('tool-tasks')
|
||||
@@ -181,6 +174,7 @@ export function apply(ctx: Context, config: Config): void {
|
||||
wait: { type: 'boolean', description: 'Block until the task reaches a terminal status or the timeout expires. A timed-out wait returns [status: running] and leaves the task alive.' },
|
||||
timeout_ms: { type: 'number', description: 'Max wait in milliseconds (only meaningful with wait: true). Defaults to the configured wait timeout; capped by the configured maximum.' },
|
||||
},
|
||||
finalizeContent: finalizeTaskContent,
|
||||
async execute(args, exec) {
|
||||
const id = validateTaskId(args.task_id)
|
||||
if (args.wait === true) {
|
||||
@@ -224,6 +218,7 @@ export function apply(ctx: Context, config: Config): void {
|
||||
task_id: { type: 'string', required: true, description: 'Task id returned by the tool that started the background work.' },
|
||||
reason: { type: 'string', description: 'Optional short reason, recorded in the log and forwarded to the task.' },
|
||||
},
|
||||
finalizeContent: finalizeTaskContent,
|
||||
execute(args, exec) {
|
||||
const id = validateTaskId(args.task_id)
|
||||
const snapshot = ctx.tasks.get(id, exec.agent)
|
||||
|
||||
@@ -162,19 +162,27 @@ describe('task_output', () => {
|
||||
expect(text(result)).toContain('[result truncated]')
|
||||
})
|
||||
|
||||
it('captures producer limits before pre- and around-execute policy', async () => {
|
||||
it('bounds pre-, around-, and post-execute policy outcomes and failures', async () => {
|
||||
const { ctx } = await setup()
|
||||
ctx.tasks.start(producer({ outputLimitBytes: 64 }).spec)
|
||||
ctx.tasks.start(producer({ outputLimitBytes: 64 }).spec)
|
||||
for (let index = 0; index < 5; index += 1) {
|
||||
ctx.tasks.start(producer({ outputLimitBytes: 64 }).spec)
|
||||
}
|
||||
ctx.on('tools/pre-execute', async (exec, next) => {
|
||||
const taskId = (exec.arguments as { task_id?: unknown }).task_id
|
||||
return taskId === 'bash-1' ? { kind: 'deny', reason: 'd'.repeat(1_000) } : next()
|
||||
if (taskId === 'bash-1') return { kind: 'deny', reason: 'd'.repeat(1_000) }
|
||||
if (taskId === 'bash-3') throw new Error(`pre failed: ${'p'.repeat(1_000)}`)
|
||||
return next()
|
||||
})
|
||||
ctx.on('tools/execute', async (exec, next) => {
|
||||
const taskId = (exec.arguments as { task_id?: unknown }).task_id
|
||||
return taskId === 'bash-2'
|
||||
? { content: [{ type: 'text', text: 'a'.repeat(1_000) }], isError: false }
|
||||
: next()
|
||||
if (taskId === 'bash-2') return { content: [{ type: 'text', text: 'a'.repeat(1_000) }], isError: false }
|
||||
if (taskId === 'bash-4') throw new Error(`around failed: ${'e'.repeat(1_000)}`)
|
||||
return next()
|
||||
})
|
||||
ctx.on('tools/post-execute', async (exec, _result, next) => {
|
||||
const taskId = (exec.arguments as { task_id?: unknown }).task_id
|
||||
if (taskId === 'bash-5') throw new Error(`post failed: ${'o'.repeat(1_000)}`)
|
||||
return next()
|
||||
})
|
||||
|
||||
const denied = await call(ctx, 'task_output', { task_id: 'bash-1' })
|
||||
@@ -186,6 +194,17 @@ describe('task_output', () => {
|
||||
expect(shortCircuited.isError).toBe(false)
|
||||
expect(Buffer.byteLength(text(shortCircuited))).toBeLessThanOrEqual(64)
|
||||
expect(text(shortCircuited)).toContain('[result truncated]')
|
||||
|
||||
const failures = [
|
||||
await call(ctx, 'task_output', { task_id: 'bash-3' }),
|
||||
await call(ctx, 'task_output', { task_id: 'bash-4' }),
|
||||
await call(ctx, 'task_output', { task_id: 'bash-5' }),
|
||||
]
|
||||
for (const failure of failures) {
|
||||
expect(failure.isError).toBe(true)
|
||||
expect(Buffer.byteLength(text(failure))).toBeLessThanOrEqual(64)
|
||||
expect(text(failure)).toContain('[result truncated]')
|
||||
}
|
||||
})
|
||||
|
||||
it('wait: true blocks until settlement and reports the terminal state', async () => {
|
||||
|
||||
Reference in New Issue
Block a user