fix(pty): close final review gaps

This commit is contained in:
Tianyi Cui
2026-07-23 02:53:43 +08:00
parent db68a90f41
commit 2821826e2c
35 changed files with 271 additions and 114 deletions

View File

@@ -10,7 +10,7 @@ The model-facing control surface for `ctx.tasks`: three kind-independent tools,
All three use generic ACP cards: `read` for output and list, `execute` for kill.
When a producer supplies `outputLimitBytes`, `task_output`, terminal `task_kill`, and completion notices cap the complete UTF-8 result after adding status or notice text. Reads retain the output tail and control suffix when they fit; a bounded completion notice instead reserves `background task <id>` and the `task_output` collection instruction before spending remaining bytes on its variable kind, label, status, and detail. An outer pre/post-execute pair captures the caller-visible task before policy and applies its producer cap to single-text denials, around-dispatch short-circuits, normalized task-control failures, replacements, and blocks; structured multi-block policy results retain their shape. An existing producer truncation marker is reused rather than duplicated. Producers that omit the field retain the existing unbounded control-surface behavior.
When a producer supplies `outputLimitBytes`, `task_output`, terminal `task_kill`, and completion notices cap the complete UTF-8 result after adding status or notice text. Reads retain the output tail and control suffix when they fit; a bounded completion notice instead reserves `background task <id>` and the `task_output` collection instruction before spending remaining bytes on its variable kind, label, status, and detail. A prepended pre-execute listener captures the caller-visible task before policy, and each task-control definition's final-content callback applies its producer cap to single-text denials, short-circuits, normalized tool or pipeline failures, replacements, and blocks; structured multi-block policy results retain their shape. An existing producer truncation marker is reused rather than duplicated. Producers that omit the field retain the existing unbounded control-surface behavior.
## Completion notices

View File

@@ -11,7 +11,7 @@ import z from 'schemastery'
import type { ContentBlock } from '@deepseek-ai/dsh-llm'
import { TextRetainer } from '@deepseek-ai/dsh-retention'
import { defineTool } from '@deepseek-ai/dsh-tools'
import type { GenericCallView, PostToolDecision, ToolExecution } from '@deepseek-ai/dsh-tools'
import type { GenericCallView, ToolDefinition, ToolExecution } from '@deepseek-ai/dsh-tools'
import { TaskId } from '@deepseek-ai/dsh-tasks'
import type { TaskSnapshot } from '@deepseek-ai/dsh-tasks'
import type {} from '@deepseek-ai/dsh-system-prompt'
@@ -128,18 +128,11 @@ export function apply(ctx: Context, config: Config): void {
if (maxBytes !== undefined) outputLimits.set(exec, maxBytes)
return next()
}, { prepend: true })
ctx.on('tools/post-execute', async (exec, result, next): Promise<PostToolDecision> => {
const decision = await next()
const maxBytes = outputLimits.get(exec)
const finalizeTaskContent: NonNullable<ToolDefinition['finalizeContent']> = (exec, result) => {
const maxBytes = outputLimits.get(exec) ?? visibleOutputLimit(ctx, exec)
outputLimits.delete(exec)
if (maxBytes === undefined) return decision
const content = decision.kind === 'block' ? decision.feedback : decision.content ?? result.content
const bounded = boundSingleText(content, maxBytes)
if (bounded === undefined) return decision
return decision.kind === 'block'
? { ...decision, feedback: bounded }
: { ...decision, content: bounded }
}, { prepend: true })
return maxBytes === undefined ? undefined : boundSingleText(result.content, maxBytes)
}
// Producers may start work only while a control surface is attached.
ctx.tasks.attachSurface('tool-tasks')
@@ -181,6 +174,7 @@ export function apply(ctx: Context, config: Config): void {
wait: { type: 'boolean', description: 'Block until the task reaches a terminal status or the timeout expires. A timed-out wait returns [status: running] and leaves the task alive.' },
timeout_ms: { type: 'number', description: 'Max wait in milliseconds (only meaningful with wait: true). Defaults to the configured wait timeout; capped by the configured maximum.' },
},
finalizeContent: finalizeTaskContent,
async execute(args, exec) {
const id = validateTaskId(args.task_id)
if (args.wait === true) {
@@ -224,6 +218,7 @@ export function apply(ctx: Context, config: Config): void {
task_id: { type: 'string', required: true, description: 'Task id returned by the tool that started the background work.' },
reason: { type: 'string', description: 'Optional short reason, recorded in the log and forwarded to the task.' },
},
finalizeContent: finalizeTaskContent,
execute(args, exec) {
const id = validateTaskId(args.task_id)
const snapshot = ctx.tasks.get(id, exec.agent)

View File

@@ -162,19 +162,27 @@ describe('task_output', () => {
expect(text(result)).toContain('[result truncated]')
})
it('captures producer limits before pre- and around-execute policy', async () => {
it('bounds pre-, around-, and post-execute policy outcomes and failures', async () => {
const { ctx } = await setup()
ctx.tasks.start(producer({ outputLimitBytes: 64 }).spec)
ctx.tasks.start(producer({ outputLimitBytes: 64 }).spec)
for (let index = 0; index < 5; index += 1) {
ctx.tasks.start(producer({ outputLimitBytes: 64 }).spec)
}
ctx.on('tools/pre-execute', async (exec, next) => {
const taskId = (exec.arguments as { task_id?: unknown }).task_id
return taskId === 'bash-1' ? { kind: 'deny', reason: 'd'.repeat(1_000) } : next()
if (taskId === 'bash-1') return { kind: 'deny', reason: 'd'.repeat(1_000) }
if (taskId === 'bash-3') throw new Error(`pre failed: ${'p'.repeat(1_000)}`)
return next()
})
ctx.on('tools/execute', async (exec, next) => {
const taskId = (exec.arguments as { task_id?: unknown }).task_id
return taskId === 'bash-2'
? { content: [{ type: 'text', text: 'a'.repeat(1_000) }], isError: false }
: next()
if (taskId === 'bash-2') return { content: [{ type: 'text', text: 'a'.repeat(1_000) }], isError: false }
if (taskId === 'bash-4') throw new Error(`around failed: ${'e'.repeat(1_000)}`)
return next()
})
ctx.on('tools/post-execute', async (exec, _result, next) => {
const taskId = (exec.arguments as { task_id?: unknown }).task_id
if (taskId === 'bash-5') throw new Error(`post failed: ${'o'.repeat(1_000)}`)
return next()
})
const denied = await call(ctx, 'task_output', { task_id: 'bash-1' })
@@ -186,6 +194,17 @@ describe('task_output', () => {
expect(shortCircuited.isError).toBe(false)
expect(Buffer.byteLength(text(shortCircuited))).toBeLessThanOrEqual(64)
expect(text(shortCircuited)).toContain('[result truncated]')
const failures = [
await call(ctx, 'task_output', { task_id: 'bash-3' }),
await call(ctx, 'task_output', { task_id: 'bash-4' }),
await call(ctx, 'task_output', { task_id: 'bash-5' }),
]
for (const failure of failures) {
expect(failure.isError).toBe(true)
expect(Buffer.byteLength(text(failure))).toBeLessThanOrEqual(64)
expect(text(failure)).toContain('[result truncated]')
}
})
it('wait: true blocks until settlement and reports the terminal state', async () => {