From 7803c38824f6617536a34b918e00fca0632c001e Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Thu, 18 Jun 2026 09:01:36 +0800 Subject: [PATCH 1/3] feat(acp): tool-owned tool-call UI presentation (title/command/output) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In Zed the tool-call card showed only "bash" — the bare tool name — instead of what the command does. Fix it by letting each TOOL own how its calls render, rather than the bridge special-casing names. dsh-tools: add an optional two-state presentation seam to ToolDefinition / defineTool — `presentCall(args)` (pending: title, kind, rawInput) and `presentResult(args, result)` (completed: title?, content?). Provider-neutral `ToolCallKind`/`ToolCallPresentation`/`ToolResultPresentation` vocabulary so tools never depend on ACP. defineTool soft-validates args (display runs on log replay, so a malformed/old shape returns undefined instead of throwing). dsh-tool-bash: bash declares presentCall (model `description` → title, exact `command` → rawInput, kind execute) and presentResult (wrap output in a fenced ```console block — a UI-only affordance kept out of the model-facing result); bash_output/bash_kill present task-scoped titles. dsh-acp: inject `tools`; a per-session `ToolPresenter` looks the tool up by name and maps its neutral presentation to the ACP tool_call/tool_call_update wire shape, with a generic fallback (title = name) for tools that declare nothing. Because the `tool/result` event carries only {callId, content, isError}, the presenter keeps a small bridge-local map of ONLY in-flight calls' (name, args), keyed by callId and removed as each result is presented — no event-schema or core change. Replay uses a throwaway presenter so loaded sessions render identically to live ones. Tests: dsh-tools defineTool presenters (typed args, soft-validate), tool-bash bash/bash_output/bash_kill presenters, acp ToolPresenter (tool-owned mapping, unknown-callId fallback, in-flight-only map), and an end-to-end turn through the bridge. The key-gated e2e now asserts a real bash call's title is the model description (not "bash") and rawInput is the command — verified against the real DeepSeek model. The test harness derives its inject from the bridge's exported `inject` so it can't drift again. --- docs/module-graph.md | 11 +- examples/acp-agent/tests/acp.e2e.ts | 17 ++- packages/acp/README.md | 8 +- packages/acp/package.json | 1 + packages/acp/src/index.ts | 127 +++++++++++++++++++-- packages/acp/tests/harness.ts | 6 +- packages/acp/tests/stream-update.spec.ts | 136 ++++++++++++++++++++++- packages/acp/tests/turns.spec.ts | 36 ++++++ packages/acp/tsconfig.json | 1 + packages/tool-bash/README.md | 4 + packages/tool-bash/src/index.ts | 44 ++++++++ packages/tool-bash/tests/tools.spec.ts | 55 +++++++++ packages/tools/README.md | 36 +++++- packages/tools/src/index.ts | 78 +++++++++++++ packages/tools/src/schema.ts | 41 ++++++- packages/tools/tests/tools.spec.ts | 51 +++++++++ 16 files changed, 626 insertions(+), 26 deletions(-) diff --git a/docs/module-graph.md b/docs/module-graph.md index a99e6cb071..5110178e0e 100644 --- a/docs/module-graph.md +++ b/docs/module-graph.md @@ -15,10 +15,6 @@ graph TD agent --> llm agent --> session session-persistence --> session - acp --> agent - acp --> llm - acp --> session - acp --> session-persistence invariants --> agent invariants --> llm invariants --> session @@ -29,6 +25,11 @@ graph TD tools --> agent tools --> llm tools --> system-prompt + acp --> agent + acp --> llm + acp --> session + acp --> session-persistence + acp --> tools agent-loop --> agent agent-loop --> llm agent-loop --> session @@ -52,10 +53,10 @@ graph TD | `system-prompt` | `llm` | | `agent` | `llm`, `session` | | `session-persistence` | `session` | -| `acp` | `agent`, `llm`, `session`, `session-persistence` | | `invariants` | `agent`, `llm`, `session` | | `session-persistence-jsonl` | `session`, `session-persistence` | | `session-persistence-sqlite` | `session`, `session-persistence` | | `tools` | `agent`, `llm`, `system-prompt` | +| `acp` | `agent`, `llm`, `session`, `session-persistence`, `tools` | | `agent-loop` | `agent`, `llm`, `session`, `session-persistence`, `system-prompt`, `tools` | | `tool-bash` | `agent`, `bash`, `llm`, `tools` | diff --git a/examples/acp-agent/tests/acp.e2e.ts b/examples/acp-agent/tests/acp.e2e.ts index 35bef79e14..caab2b0f4a 100644 --- a/examples/acp-agent/tests/acp.e2e.ts +++ b/examples/acp-agent/tests/acp.e2e.ts @@ -174,6 +174,21 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('acp-agent e2e: real prompt over expect(proof).toContain('ACP_OK') // And the client saw tool-call activity stream through. - expect(updates.some(u => u.sessionUpdate === 'tool_call')).toBe(true) + const toolCalls = updates.filter(u => u.sessionUpdate === 'tool_call') + expect(toolCalls.length).toBeGreaterThan(0) + + // Tool-call UI quality (the tool owns its presentation): the bash tool's + // `presentCall` sets the title to the model's human-readable `description` + // and the `rawInput` to the exact command — NOT the bare tool name "bash". + // A `bash` call must therefore carry an execute kind, a non-"bash" title, + // and a string rawInput (the command). `toolCalls` is already narrowed to + // the `tool_call` shape by the filter above, so these fields are reachable. + const bashCall = toolCalls.find(u => u.kind === 'execute') + expect(bashCall).toBeDefined() + if (bashCall === undefined) throw new Error('expected an execute tool_call') + expect(typeof bashCall.title).toBe('string') + expect(bashCall.title.length).toBeGreaterThan(0) + expect(bashCall.title).not.toBe('bash') // the old, unhelpful title + expect(typeof bashCall.rawInput).toBe('string') // the exact command }, 180_000) }) diff --git a/packages/acp/README.md b/packages/acp/README.md index e528acdb45..f515ae8293 100644 --- a/packages/acp/README.md +++ b/packages/acp/README.md @@ -28,7 +28,7 @@ It is a **client-driver / UI plugin**, the structured analogue of the readline ` | `session/load` | `ctx.agents.resume(...)` | replays the persisted event log to the client as `session/update` — the USER side (`user/message` → `user_message_chunk`), assistant text/reasoning (`assistant/chunk`), and tool calls/results (`tool/call` + `tool/result`). Re-loading an already-live id is rejected; the id's load slot is reserved (`loadingIds`) BEFORE the async resume so a pipelined load of the SAME id can't leak a second agent (distinct ids load concurrently). The resumed session keeps its PERSISTED header `cwd`, so its bash tools run in the original workspace; the requested `cwd` only needs to be absolute. After the async resume a `closed` re-check refuses to install a record if the bridge tore down mid-load | | `session/prompt` | `agent.send()` | text-only; rejects image/audio and empty prompts; one in-flight prompt PER session (independent); settles on the OWNING turn's end (a turn that ends in `error` rejects the RPC) | | `session/cancel` | `agent.abort()` | aborts a running step + settles the prompt `cancelled` for ONLY that session — a cancel never touches another session's stream or prompt (see limitation below) | -| `session/update` | `session/event` | `agent_message_chunk` (text-delta), `agent_thought_chunk` (reasoning-delta), `user_message_chunk` (load replay), `tool_call`/`tool_call_update` | +| `session/update` | `session/event` | `agent_message_chunk` (text-delta), `agent_thought_chunk` (reasoning-delta), `user_message_chunk` (load replay), `tool_call`/`tool_call_update` (title/kind/rawInput/content owned by the TOOL via `presentCall`/`presentResult` — see Tool-call presentation) | ## Multi-session (RFC 011) @@ -40,6 +40,12 @@ Background-task isolation rides on `dsh-tool-bash`: bash task ids are global and Each session runs in its own workspace, recorded as the session's `SessionHeader.cwd`. On `session/new` the (absolute) request `cwd` becomes that header cwd; on `session/load` the resumed session keeps its PERSISTED header cwd (the request `cwd` is only shape-checked — it does not override the stored one), and a load whose persisted session has no absolute cwd is REJECTED up front via a metadata-only `list()` check, BEFORE resume constructs an agent (else bash would silently fall back to the server's launch dir, and a post-resume reject would leak the registered agent). `dsh-tool-bash` then defaults the bash workdir to the calling agent's `session.header.cwd` (an explicit model `workdir` still wins; a relative one resolves against the session cwd; with no session cwd the executor falls back to its own config / `process.cwd()`). So the server no longer has to be launched in the workspace — an editor can open any project folder, and N sessions over one connection can each target a different directory. (`additionalDirectories` is still rejected: widening the tool/filesystem scope beyond the single cwd is a separate sandbox concern.) +## Tool-call presentation + +How a tool call renders in the editor is owned by the TOOL, not the bridge — the bridge never special-cases tool names. Each tool may declare `presentCall(args)` (pending state: a human-readable `title`, a `kind` for the icon, and the salient `rawInput` to show in a detail view) and `presentResult(args, result)` (completed state: an optional replacement `title` and reformatted `content`) on its `dsh-tools` definition. The bridge looks the definition up by name in `ctx.tools` and maps the neutral `ToolCallPresentation`/`ToolResultPresentation` to the ACP `tool_call`/`tool_call_update` wire shapes. A tool that declares neither gets a generic fallback (title = tool name, raw parsed args as `rawInput`, kind inferred from the name). For example `dsh-tool-bash` makes the model-written one-line `description` the title ("List files in the current directory"), the exact `command` the `rawInput`, `kind: 'execute'`, and wraps the completed output in a fenced ` ```console ` block. + +The `tool/result` session event carries only `{ callId, content, isError }` — not the tool name or args — so to call a tool's `presentResult` the bridge keeps a small per-session map from `callId` to the in-flight call's `(name, args)`, populated on `tool/call` and removed as each result is presented (it holds only currently-in-flight calls, never finished ones). This is bridge-local state — NOT a change to the event schema or a core service. The map lives on the `SessionRecord`, so two concurrent sessions never cross their in-flight tool state; a `session/load` replay uses a throwaway presenter that pairs each `tool/call` with its `tool/result` as the log replays in order, so replayed tool cards render identically to live ones. + ## Settle-exactly-once A `session/prompt` resolves (or rejects) exactly once, keyed off the canonical session log (the `session/event` stream), NOT the `agent/turn-start`/`agent/turn-end` events. One listener captures the prompt's owning turn from the log's `turn/start` and settles on the matching `turn/end` — the one signal that always fires (`closeTurn` appends it unconditionally, even when a boundary emit throws and the `agent/turn-end` EVENT is skipped). A prompt settles only on ITS OWN turn (`inflight.turn === turn/end.turn`), so a stale `turn/end` for a previously-cancelled turn whose end arrives late can never settle the wrong prompt. A turn that ends `error` REJECTS the RPC with an internal error carrying the failure message (ACP has no error stop reason); every other reason resolves via the codec. As a fallback, when the agent settles to `idle`/`disposed` with a prompt still pending — e.g. a `session/event` listener registered before the bridge threw and starved the bridge's listener — an `agent/status` handler reconciles the prompt from the log (the owning turn's `turn/end`, or `cancelled` if the turn was torn down without one). An empty/whitespace prompt is rejected up front — it would queue no work, so no turn would start and the RPC would hang. diff --git a/packages/acp/package.json b/packages/acp/package.json index 669c03043d..9cac888518 100644 --- a/packages/acp/package.json +++ b/packages/acp/package.json @@ -29,6 +29,7 @@ "@deepseek-ai/dsh-llm": "^0.0.1", "@deepseek-ai/dsh-session": "^0.0.1", "@deepseek-ai/dsh-session-persistence": "^0.0.1", + "@deepseek-ai/dsh-tools": "^0.0.1", "cordis": "^4.0.0-rc.6" }, "devDependencies": { diff --git a/packages/acp/src/index.ts b/packages/acp/src/index.ts index 18f3adb670..793685072b 100644 --- a/packages/acp/src/index.ts +++ b/packages/acp/src/index.ts @@ -60,6 +60,7 @@ import { import type { ContentBlock } from '@deepseek-ai/dsh-llm' import type { Agent, AgentStatus } from '@deepseek-ai/dsh-agent' import type { SessionEvent } from '@deepseek-ai/dsh-session' +import type { ToolCallKind, ToolRegistry } from '@deepseek-ai/dsh-tools' // Side-effect type import: declaration-merges `ctx.sessionPersistence` onto // Context (the bridge injects it and reads `list()` for load cwd validation). import type {} from '@deepseek-ai/dsh-session-persistence' @@ -73,8 +74,10 @@ import { export const name = 'acp' // The bridge programs against the interface packages only (architecture rule: // plugins never depend on dsh-agent-loop). `sessionPersistence` is required -// because `initialize` advertises `loadSession: true`. -export const inject = ['agents', 'sessions', 'sessionPersistence'] +// because `initialize` advertises `loadSession: true`. `tools` lets a tool own +// how its calls render (`presentCall`/`presentResult`); the bridge looks up the +// definition by name and falls back to a generic presentation when absent. +export const inject = ['agents', 'sessions', 'sessionPersistence', 'tools'] /** * Build an ACP "invalid params" error whose human detail rides in the message. @@ -131,6 +134,13 @@ export const Config: Schema = Schema.object({ interface SessionRecord { sessionId: string agent: Agent + /** + * Resolves tool-owned presentation for THIS session's tool calls and remembers + * each in-flight call's `(name, args)` so the matching `tool/result` can find + * its tool. Per-session so two concurrent sessions never cross their in-flight + * tool state. + */ + presenter: ToolPresenter /** * The in-flight `session/prompt`, or `undefined` when none is pending. A * prompt resolves with a {@link StopReason} or rejects with an Error (a @@ -182,6 +192,7 @@ export function apply(ctx: Context, config: AcpConfig): void { const agents = ctx.agents const sessionPersistence = ctx.sessionPersistence const logger = ctx.logger + const tools = ctx.tools // Live sessions keyed by id (RFC 011 multi-session), plus an agent→sessionId // reverse map so `agent/*` events (which carry only the Agent) demux in O(1). @@ -270,7 +281,7 @@ export function apply(ctx: Context, config: AcpConfig): void { ctx.on('session/event', (session, event: SessionEvent) => { const rec = sessions.get(session.header.id) if (rec === undefined) return - streamSessionEventUpdate(rec.sessionId, event, notify) + streamSessionEventUpdate(rec.sessionId, event, notify, rec.presenter) const inflight = rec.inflight if (inflight === undefined) return if (event.type === 'turn/start') { @@ -395,7 +406,7 @@ export function apply(ctx: Context, config: AcpConfig): void { agentOptions: agentOptions(config), }) bySession.set(agent, sessionId) - sessions.set(sessionId, { sessionId, agent, inflight: undefined }) + sessions.set(sessionId, { sessionId, agent, presenter: new ToolPresenter(tools), inflight: undefined }) return Promise.resolve({ sessionId }) }, @@ -448,14 +459,18 @@ export function apply(ctx: Context, config: AcpConfig): void { throw invalidParams('connection closed during session/load') } bySession.set(agent, params.sessionId) - sessions.set(params.sessionId, { sessionId: params.sessionId, agent, inflight: undefined }) + const record: SessionRecord = { sessionId: params.sessionId, agent, presenter: new ToolPresenter(tools), inflight: undefined } + sessions.set(params.sessionId, record) // Replay the persisted event log to the client as session/update. Use // the raw event log (NOT deriveMessages, which drops assistant/chunk // and trace events): RFC 010's load contract reconstructs the streamed // turns — user prompts (user/message → user_message_chunk), assistant - // text and reasoning (assistant/chunk), and tool calls/results. + // text and reasoning (assistant/chunk), and tool calls/results. The + // record's presenter pairs each tool/call with its tool/result as the + // log replays in order, so the replayed tool cards render identically + // to the live ones. for (const event of agent.session.events) { - streamSessionEventUpdate(params.sessionId, event, notify) + streamSessionEventUpdate(params.sessionId, event, notify, record.presenter) } return {} } finally { @@ -659,6 +674,14 @@ function validateWorkspaceParams(params: { cwd: string; additionalDirectories?: * - `tool/call` → `tool_call` (pending) * - `tool/result` → `tool_call_update` (completed/failed) * + * Tool-call presentation (title/kind/rawInput, and the completed-state content) + * is owned by each TOOL via `presentCall`/`presentResult` — the bridge never + * special-cases tool names. `presenter` resolves those from the tool registry + * and remembers each call's `(name, args)` so the completed `tool/result` (which + * carries neither) can find its tool. A {@link nullToolPresenter} gives the + * generic fallback (title = tool name, raw args as input) when no registry is + * available (e.g. pure translator tests). + * * Other event types (turn/step boundaries, context/message, usage, …) produce * no client update. */ @@ -666,6 +689,7 @@ export function streamSessionEventUpdate( sessionId: string, event: SessionEvent, notify: (notification: SessionNotification) => void, + presenter: Pick = nullToolPresenter, ): void { switch (event.type) { case 'assistant/chunk': { @@ -690,27 +714,30 @@ export function streamSessionEventUpdate( return } case 'tool/call': { + const present = presenter.call(event.data.callId, event.data.name, event.data.arguments) notify({ sessionId, update: { sessionUpdate: 'tool_call', toolCallId: event.data.callId, - title: event.data.name, - kind: toolKindFor(event.data.name), + title: present.title, + kind: present.kind, status: 'in_progress', - rawInput: parseToolArguments(event.data.arguments), + ...present.rawInput !== undefined ? { rawInput: present.rawInput } : {}, }, }) return } case 'tool/result': { + const present = presenter.result(event.data.callId, event.data.content, event.data.isError) notify({ sessionId, update: { sessionUpdate: 'tool_call_update', toolCallId: event.data.callId, status: event.data.isError ? 'failed' : 'completed', - content: toolResultContent(event.data.content), + content: toolResultContent(present.content), + ...present.title !== undefined ? { title: present.title } : {}, }, }) return @@ -722,8 +749,84 @@ export function streamSessionEventUpdate( } } +/** + * Resolved pending-state presentation the bridge feeds into a `tool_call` + * update: a title is always present (tool name when the tool gives none), `kind` + * and `rawInput` are optional. + */ +interface ResolvedCallPresentation { + title: string + kind: ToolCallKind + rawInput?: unknown +} + +/** Resolved completed-state presentation fed into a `tool_call_update`. */ +interface ResolvedResultPresentation { + /** UI content for the result (harness blocks; the tool may reformat, else the raw result). */ + content: ContentBlock[] + /** Optional replacement title for the completed call. */ + title?: string +} + +/** + * Resolves tool-owned presentation for a session's tool-call events. A tool + * declares `presentCall`/`presentResult` (see `dsh-tools`); this looks them up + * by name in the registry and applies the generic fallback when a tool defines + * neither. + * + * The `tool/result` session event carries only `{ callId, content, isError }` — + * NOT the tool name or args — so to call a tool's `presentResult` (which needs + * both), the presenter remembers each `tool/call`'s `{ name, args }` keyed by + * callId and looks it up on the matching result. The map is bridge-LOCAL (not a + * change to the event schema or a core service): one presenter per live session + * (and a throwaway per `session/load` replay), entries removed as each result + * arrives, so it holds only the currently-in-flight calls. + */ +export class ToolPresenter { + private readonly pending = new Map() + + constructor(private readonly tools: Pick) {} + + /** Pending-state presentation for a `tool/call`; remembers `(name, args)` for the matching result. */ + call(callId: string, name: string, argsJson: string): ResolvedCallPresentation { + const args = parseToolArguments(argsJson) + this.pending.set(callId, { name, args }) + const present = this.tools.get(name)?.presentCall?.(args) + if (present === undefined) { + // No tool-owned presentation: fall back to the tool name as the title and + // the full parsed args as the raw input (the pre-seam behavior). + return { title: name, kind: toolKindFor(name), rawInput: args } + } + return { title: present.title, kind: present.kind ?? 'other', rawInput: present.rawInput } + } + + /** Completed-state presentation for a `tool/result`; consumes the remembered `(name, args)`. */ + result(callId: string, content: ContentBlock[], isError: boolean): ResolvedResultPresentation { + const call = this.pending.get(callId) + this.pending.delete(callId) + const present = call !== undefined + ? this.tools.get(call.name)?.presentResult?.(call.args, { content, isError }) + : undefined + if (present === undefined) return { content } + return { + content: present.content ?? content, + ...present.title !== undefined ? { title: present.title } : {}, + } + } +} + +/** + * The no-op presenter used when no tool registry is available (e.g. the pure + * translator tests): every tool gets the generic fallback presentation, and + * results pass their raw content through unchanged. + */ +export const nullToolPresenter: Pick = { + call: (_callId, name, argsJson) => ({ title: name, kind: toolKindFor(name), rawInput: parseToolArguments(argsJson) }), + result: (_callId, content) => ({ content }), +} + /** Map a harness tool name to an ACP ToolKind (best-effort; default `other`). */ -function toolKindFor(name: string): 'read' | 'edit' | 'delete' | 'move' | 'search' | 'execute' | 'fetch' | 'other' { +function toolKindFor(name: string): ToolCallKind { if (name === 'bash' || name === 'bash_output' || name === 'bash_kill') return 'execute' if (name === 'read' || name.startsWith('read')) return 'read' if (name === 'write' || name === 'edit' || name.startsWith('edit')) return 'edit' diff --git a/packages/acp/tests/harness.ts b/packages/acp/tests/harness.ts index bee16c46eb..27335cfcff 100644 --- a/packages/acp/tests/harness.ts +++ b/packages/acp/tests/harness.ts @@ -234,7 +234,11 @@ export async function makeBridgeHarness(options: { // tears down JUST the bridge (its listeners + effect) for the HMR test. harness.acpFiber = await ctx.plugin({ name: 'acp-test', - inject: ['agents', 'sessions', 'sessionPersistence'], + // Use the bridge's REAL exported `inject` so this never drifts from the + // plugin's actual dependency list (adding a service to the bridge must not + // require editing the harness — a hardcoded list silently broke when `tools` + // was added). The bridge programs against the interface packages only. + inject: [...AcpPlugin.inject], apply: (inner: Context) => { AcpPlugin.apply(inner, cfg) }, }) harness.client = new ClientSideConnection(makeClient, clientStream) diff --git a/packages/acp/tests/stream-update.spec.ts b/packages/acp/tests/stream-update.spec.ts index c37de8b344..4328b5fb6b 100644 --- a/packages/acp/tests/stream-update.spec.ts +++ b/packages/acp/tests/stream-update.spec.ts @@ -2,15 +2,22 @@ import { describe, expect, it } from 'vitest' import { CallId } from '@deepseek-ai/dsh-llm' import type { SessionEvent } from '@deepseek-ai/dsh-session' import type { SessionNotification } from '@agentclientprotocol/sdk' -import { streamSessionEventUpdate, agentOptions } from '../src/index.ts' +import type { ToolDefinition, ToolRegistry } from '@deepseek-ai/dsh-tools' +import { streamSessionEventUpdate, agentOptions, ToolPresenter } from '../src/index.ts' -/** Collect the updates a single event produces. */ +/** Collect the updates a single event produces (no presenter → generic fallback). */ function updatesFor(event: SessionEvent): SessionNotification['update'][] { const out: SessionNotification['update'][] = [] streamSessionEventUpdate('s1', event, n => out.push(n.update)) return out } +/** A tiny tool registry stub exposing just `get` for {@link ToolPresenter}. */ +function registryOf(...tools: ToolDefinition[]): Pick { + const map = new Map(tools.map(t => [t.name, t])) + return { get: name => map.get(name) } +} + function evt(type: T, data: Extract['data']): SessionEvent { return { type, seq: 0, time: 0, data } as SessionEvent } @@ -31,7 +38,7 @@ describe('streamSessionEventUpdate', () => { .toEqual([]) }) - it('maps tool/call to an in_progress tool_call with inferred kind and parsed rawInput', () => { + it('maps tool/call to an in_progress tool_call with inferred kind and parsed rawInput (generic fallback, no presenter)', () => { const updates = updatesFor(evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'bash', arguments: '{"command":"ls"}' })) expect(updates).toEqual([{ sessionUpdate: 'tool_call', @@ -99,6 +106,129 @@ describe('streamSessionEventUpdate', () => { }) }) +describe('ToolPresenter (tool-owned presentation via the tool registry)', () => { + /** A tool whose presentCall/presentResult mirror what tool-bash declares. */ + const bashLike: ToolDefinition = { + name: 'bash', + description: 'run a command', + parameters: {}, + execute: async () => [], + presentCall: (args: unknown) => { + const a = args as { command: string; description: string } + return { title: a.description, kind: 'execute', rawInput: a.command } + }, + presentResult: (_args: unknown, result: { content: { type: string }[] }) => ({ + content: [{ type: 'text', text: `wrapped:${result.content.length}` }], + }), + } + + function updatesWith(presenter: ToolPresenter, ...events: SessionEvent[]): SessionNotification['update'][] { + const out: SessionNotification['update'][] = [] + for (const event of events) streamSessionEventUpdate('s1', event, n => out.push(n.update), presenter) + return out + } + + it('tool/call uses the tool: description→title, command→rawInput, tool kind', () => { + const presenter = new ToolPresenter(registryOf(bashLike)) + const [update] = updatesWith(presenter, evt('tool/call', { + turn: 1, step: 1, callId: CallId('c1'), name: 'bash', + arguments: JSON.stringify({ command: 'ls -la', description: 'List files' }), + })) + expect(update).toEqual({ + sessionUpdate: 'tool_call', + toolCallId: 'c1', + title: 'List files', + kind: 'execute', + status: 'in_progress', + rawInput: 'ls -la', + }) + }) + + it('tool/result uses the tool to reformat content (resolved by the remembered tool/call)', () => { + const presenter = new ToolPresenter(registryOf(bashLike)) + const updates = updatesWith( + presenter, + evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'bash', arguments: JSON.stringify({ command: 'x', description: 'd' }) }), + evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'out' }], isError: false }), + ) + expect(updates[1]).toEqual({ + sessionUpdate: 'tool_call_update', + toolCallId: 'c1', + status: 'completed', + content: [{ type: 'content', content: { type: 'text', text: 'wrapped:1' } }], + }) + }) + + it('a result with NO preceding call (unknown callId) falls back to the raw content', () => { + const presenter = new ToolPresenter(registryOf(bashLike)) + // No tool/call for c9 → presenter has nothing remembered → generic fallback. + const [update] = updatesWith(presenter, evt('tool/result', { + turn: 1, step: 1, callId: CallId('c9'), content: [{ type: 'text', text: 'raw' }], isError: false, + })) + expect(update).toEqual({ + sessionUpdate: 'tool_call_update', + toolCallId: 'c9', + status: 'completed', + content: [{ type: 'content', content: { type: 'text', text: 'raw' } }], + }) + }) + + it('a tool with no presentCall/presentResult gets the generic fallback (title = name)', () => { + const plain: ToolDefinition = { name: 'plain', description: 'p', parameters: {}, execute: async () => [] } + const presenter = new ToolPresenter(registryOf(plain)) + const [update] = updatesWith(presenter, evt('tool/call', { + turn: 1, step: 1, callId: CallId('c1'), name: 'plain', arguments: '{"a":1}', + })) + expect(update).toMatchObject({ title: 'plain', kind: 'other', rawInput: { a: 1 } }) + }) + + it('a presentation that omits kind/content/rawInput uses the defaults (kind other, raw result content kept)', () => { + // A minimal tool-owned presentation: presentCall returns only a title (no + // kind → defaults to `other`, no rawInput → omitted); presentResult returns + // only a title (no content → the raw result content is kept). + const minimal: ToolDefinition = { + name: 'mini', + description: 'm', + parameters: {}, + execute: async () => [], + presentCall: () => ({ title: 'Doing a thing' }), + presentResult: () => ({ title: 'Did the thing' }), + } + const presenter = new ToolPresenter(registryOf(minimal)) + const updates = updatesWith( + presenter, + evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'mini', arguments: '{}' }), + evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'kept' }], isError: false }), + ) + // No kind → 'other'; no rawInput key at all. + expect(updates[0]).toEqual({ sessionUpdate: 'tool_call', toolCallId: 'c1', title: 'Doing a thing', kind: 'other', status: 'in_progress' }) + // Title replaced; content falls back to the raw result content. + expect(updates[1]).toEqual({ + sessionUpdate: 'tool_call_update', + toolCallId: 'c1', + status: 'completed', + content: [{ type: 'content', content: { type: 'text', text: 'kept' } }], + title: 'Did the thing', + }) + }) + + it('holds ONLY in-flight calls: the callId entry is removed once its result is presented', () => { + const presenter = new ToolPresenter(registryOf(bashLike)) + updatesWith( + presenter, + evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'bash', arguments: JSON.stringify({ command: 'x', description: 'd' }) }), + evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'o' }], isError: false }), + ) + // A SECOND result for the same callId now finds nothing remembered, so it + // falls back to raw content (proving the first result consumed the entry — + // the map does not retain finished calls). + const [late] = updatesWith(presenter, evt('tool/result', { + turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'late' }], isError: false, + })) + expect(late).toMatchObject({ content: [{ type: 'content', content: { type: 'text', text: 'late' } }] }) + }) +}) + describe('agentOptions', () => { it('includes only the fields present in config', () => { expect(agentOptions({})).toEqual({}) diff --git a/packages/acp/tests/turns.spec.ts b/packages/acp/tests/turns.spec.ts index eef0f392aa..7258c590b4 100644 --- a/packages/acp/tests/turns.spec.ts +++ b/packages/acp/tests/turns.spec.ts @@ -75,6 +75,42 @@ describe('acp bridge — turn outcomes', () => { expect(callIdx).toBeLessThan(updIdx) }) + it('a tool-owned presentation flows end-to-end: presentCall sets title/rawInput, presentResult reformats output', async () => { + harness = await makeBridgeHarness({ + storageDir, + script: [toolCallResponse('c1', 'bash', { command: 'ls -la', description: 'List files' }), textResponse('done')], + }) + // A tool that declares its OWN presentation (like the real tool-bash). The + // bridge must use it — NOT the generic title=name fallback — proving the + // tool-owns-its-rendering seam works through the real session-event path. + harness.ctx.tools.register(defineTool({ + name: 'bash', + description: 'run a command', + parameters: { + command: { type: 'string', required: true }, + description: { type: 'string', required: true }, + }, + async execute() { return [{ type: 'text', text: 'a.txt\nb.txt\n' }] }, + presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), + presentResult: (_args, result) => { + const block = result.content.length === 1 ? result.content[0] : undefined + if (block === undefined || block.type !== 'text') return undefined + return { content: [{ type: 'text', text: `\`\`\`console\n${block.text.trimEnd()}\n\`\`\`` }] } + }, + })) + const sessionId = await newSession(harness) + await harness.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'list' }] }) + + const call = harness.updates.find(u => u.sessionUpdate === 'tool_call') + expect(call).toMatchObject({ toolCallId: 'c1', title: 'List files', kind: 'execute', rawInput: 'ls -la', status: 'in_progress' }) + const update = harness.updates.find(u => u.sessionUpdate === 'tool_call_update') + expect(update).toMatchObject({ + toolCallId: 'c1', + status: 'completed', + content: [{ type: 'content', content: { type: 'text', text: '```console\na.txt\nb.txt\n```' } }], + }) + }) + it('a failing tool yields a failed tool_call_update', async () => { harness = await makeBridgeHarness({ storageDir, diff --git a/packages/acp/tsconfig.json b/packages/acp/tsconfig.json index 2efad36448..83330256e3 100644 --- a/packages/acp/tsconfig.json +++ b/packages/acp/tsconfig.json @@ -12,6 +12,7 @@ { "path": "../llm" }, { "path": "../session" }, { "path": "../agent" }, + { "path": "../tools" }, { "path": "../session-persistence" } ] } diff --git a/packages/tool-bash/README.md b/packages/tool-bash/README.md index 83f84e967d..473f5a4099 100644 --- a/packages/tool-bash/README.md +++ b/packages/tool-bash/README.md @@ -32,6 +32,10 @@ Result text: stdout, then a `[stderr]` section, then status markers — `[timed The owning agent is recorded per task id at spawn and kept for the lifetime of the loaded plugin instance (it is **not** cleared on completion). `bash_output`/`bash_kill` reject a task owned by a *different* agent with `task belongs to another session` (a task started with no agent — a non-loop caller — has no owner and is open to anyone; a call with no `exec.agent` cannot access an owned task). Task ids are global and predictable, so under multi-session ACP this ownership check is the fence that stops one session's agent from reading or killing another session's background task. (`TODO(tool-bash-owner-hmr)`: an independent HMR reload of this plugin starts a fresh map, so a task spawned before the reload becomes un-owned — acceptable as HMR is dev-only and the session boundary is one user's cooperative editor; a durable fix attaches ownership to the executor/task lifetime.) +## UI presentation + +These tools own how their calls render in a UI (an editor's tool-call card) via the `dsh-tools` `presentCall`/`presentResult` seam — a UI never special-cases tool names. For `bash`: the model-written `description` is the always-visible **title** (e.g. "List files in the current directory"), the exact `command` is the **rawInput** (the verbatim command stays visible in a detail view without crowding the title), `kind` is `execute` (terminal/run treatment), and the completed output is wrapped in a fenced ` ```console ` block — a UI-only affordance, so the model-facing result text stays unfenced. `bash_output`/`bash_kill` present a task-scoped title ("Read output from background task bash-3" / "Kill background task bash-3") with the task id as rawInput. These methods are pure/display-only (they also run on `session/load` replay), and a malformed/older logged arg shape falls back to a generic presentation rather than throwing. See `packages/tools` ("Tool-owned UI presentation") and `packages/acp` ("Tool-call presentation"). + ## Background completion notices When a background task finishes, a short notice is injected into the owning agent's session (`agent.inject()`, source `{kind: 'plugin', plugin: 'tool-bash'}`). Injection is **durable context for the next model request, not a wake-up** — an idle agent stays idle until something sends a message. That's why the tool descriptions tell the model to poll with `bash_output`. diff --git a/packages/tool-bash/src/index.ts b/packages/tool-bash/src/index.ts index 9a15c6ad7a..9b4aa0819b 100644 --- a/packages/tool-bash/src/index.ts +++ b/packages/tool-bash/src/index.ts @@ -39,6 +39,8 @@ import type { Context } from 'cordis' import { isAbsolute, resolve as resolvePath } from 'node:path' import { defineTool } from '@deepseek-ai/dsh-tools' +import type { ToolCallPresentation, ToolResult, ToolResultPresentation } from '@deepseek-ai/dsh-tools' +import type { ContentBlock } from '@deepseek-ai/dsh-llm' import type { Agent } from '@deepseek-ai/dsh-agent' import type { BashRunResult, BashTask, CollectedOutput } from '@deepseek-ai/dsh-bash' @@ -124,6 +126,44 @@ export function renderResult(result: BashRunResult): string { return body + markers.join('\n') } +// --------------------------------------------------------------------------- +// UI presentation (tool-owned). These shape how a UI (e.g. the ACP bridge) +// renders a bash call's pending and completed states. They are display-only and +// pure — a UI may call them during live streaming AND a session-log replay. +// --------------------------------------------------------------------------- + +/** + * Pending-state presentation for a `bash` call: the model-written `description` + * is the always-visible title (the schema requires it precisely so a UI has a + * readable summary — "List files in the current directory"), `kind: 'execute'` + * (a terminal/run treatment), and the exact `command` is the `rawInput` so the + * verbatim command stays visible in a UI's detail view without crowding the + * title. Mirrors how Zed / the reference ACP adapters render execute tools. + */ +function presentBashCall(args: { command: string; description: string }): ToolCallPresentation { + return { title: args.description, kind: 'execute', rawInput: args.command } +} + +/** + * Completed-state presentation for a `bash` call: wrap the model-facing result + * text in a fenced ```console block so a UI renders the output monospaced as a + * terminal transcript. The model-facing `content` (what `execute` returned) is + * intentionally NOT fenced — the fences are a UI-only affordance, so they live + * here, not in `renderResult`. A non-text result (unexpected for bash) is left + * untouched by falling back to `undefined`. + */ +function presentBashResult(_args: unknown, result: ToolResult): ToolResultPresentation | undefined { + const block = result.content.length === 1 ? result.content[0] : undefined + if (block === undefined || block.type !== 'text') return undefined + const fenced: ContentBlock = { type: 'text', text: `\`\`\`console\n${block.text.replace(/\n+$/, '')}\n\`\`\`` } + return { content: [fenced] } +} + +/** Pending-state presentation for `bash_output`/`bash_kill` (background-task tools). */ +function presentTaskCall(verb: string, args: { task_id: string }): ToolCallPresentation { + return { title: `${verb} background task ${args.task_id}`, kind: 'execute', rawInput: args.task_id } +} + /** * Resolve the working directory for a bash call. Precedence: an explicit model * `workdir` wins; otherwise default to the calling agent's session cwd @@ -242,6 +282,8 @@ export function apply(ctx: Context): void { if (result.aborted) throw new Error('command aborted') return [{ type: 'text', text: renderResult(result) }] }, + presentCall: presentBashCall, + presentResult: presentBashResult, })) ctx.tools.register(defineTool({ @@ -266,6 +308,7 @@ export function apply(ctx: Context): void { text += `\n${statusLine(read.task)}` return Promise.resolve([{ type: 'text', text }]) }, + presentCall: args => presentTaskCall('Read output from', args), })) ctx.tools.register(defineTool({ @@ -283,5 +326,6 @@ export function apply(ctx: Context): void { text: killed ? `killed background task ${id}` : `task ${id} had already finished`, }]) }, + presentCall: args => presentTaskCall('Kill', args), })) } diff --git a/packages/tool-bash/tests/tools.spec.ts b/packages/tool-bash/tests/tools.spec.ts index cc25d9d9d4..8b68de213b 100644 --- a/packages/tool-bash/tests/tools.spec.ts +++ b/packages/tool-bash/tests/tools.spec.ts @@ -562,3 +562,58 @@ describe('status lines', () => { expect(text(read)).toContain('[status: completed, exit code: 0]') }) }) + +describe('tool-owned UI presentation (presentCall / presentResult)', () => { + it('bash presentCall: the model description is the title, the command is the rawInput, kind execute', async () => { + const ctx = await setup() + const present = ctx.tools.get('bash')!.presentCall!({ command: 'ls -la src', description: 'List files in src' }) + expect(present).toEqual({ title: 'List files in src', kind: 'execute', rawInput: 'ls -la src' }) + }) + + it('bash presentResult: wraps the model-facing text in a fenced console block', async () => { + const ctx = await setup() + const present = ctx.tools.get('bash')!.presentResult!( + { command: 'echo hi', description: 'echo' }, + { content: [{ type: 'text', text: 'hi\n[exit code: 0]\n\n' }], isError: false }, + ) + // Trailing blank lines are trimmed; the body is fenced as ```console. + expect(present).toEqual({ content: [{ type: 'text', text: '```console\nhi\n[exit code: 0]\n```' }] }) + }) + + it('bash presentResult: leaves a non-text (unexpected) result untouched → undefined (UI keeps raw content)', async () => { + const ctx = await setup() + const present = ctx.tools.get('bash')!.presentResult!( + { command: 'x', description: 'x' }, + { content: [{ type: 'image', url: 'https://x/y.png' }], isError: false }, + ) + expect(present).toBeUndefined() + }) + + it('bash presentResult: a result that is not exactly one block → undefined (no single text to fence)', async () => { + const ctx = await setup() + const args = { command: 'x', description: 'x' } + // Empty content (no block) and multi-block content both fall through. + expect(ctx.tools.get('bash')!.presentResult!(args, { content: [], isError: false })).toBeUndefined() + expect(ctx.tools.get('bash')!.presentResult!(args, { + content: [{ type: 'text', text: 'a' }, { type: 'text', text: 'b' }], + isError: false, + })).toBeUndefined() + }) + + it('bash_output / bash_kill presentCall: a readable task-scoped title, task id as rawInput', async () => { + const ctx = await setup() + expect(ctx.tools.get('bash_output')!.presentCall!({ task_id: 'bash-3' })) + .toEqual({ title: 'Read output from background task bash-3', kind: 'execute', rawInput: 'bash-3' }) + expect(ctx.tools.get('bash_kill')!.presentCall!({ task_id: 'bash-3' })) + .toEqual({ title: 'Kill background task bash-3', kind: 'execute', rawInput: 'bash-3' }) + }) + + it('presentCall validates softly: malformed args (missing required description) return undefined, never throw', async () => { + const ctx = await setup() + // defineTool wraps presentCall to soft-validate against the schema and fall + // back to undefined (a generic UI presentation) rather than throwing on the + // display path — it may run on replay of arbitrary logged args. The + // ToolDefinition.presentCall takes `unknown`, so a malformed shape needs no cast. + expect(ctx.tools.get('bash')?.presentCall?.({ command: 'ls' })).toBeUndefined() + }) +}) diff --git a/packages/tools/README.md b/packages/tools/README.md index 560195387d..d33e28f1e9 100644 --- a/packages/tools/README.md +++ b/packages/tools/README.md @@ -24,9 +24,10 @@ Tool registry and execution waterfall. Tool plugins register their schemas and e ### Key types -- `ToolDefinition` — `ToolSchema` + `execute(args, exec): Promise`. +- `ToolDefinition` — `ToolSchema` + `execute(args, exec): Promise`, plus optional `presentCall(args)` / `presentResult(args, result)` for tool-owned UI presentation (see below). - `ToolExecution` — one pending tool call: `{ callId, name, arguments, agent?, signal? }`. - `ToolExecutionResult` — outcome: `{ callId, content, isError, error? }`. On failure with a `HarnessError`, `error: { name, code }` carries the structured failure class alongside the model-facing text (the loop forwards it onto the `tool/result` session event for retry/sandbox plugins and replay). +- `ToolCallPresentation` / `ToolResultPresentation` — provider-neutral shapes a tool returns from `presentCall` / `presentResult` to own how a UI renders ITS calls (see "Tool-owned UI presentation"). ### Extension points @@ -67,6 +68,39 @@ A `defineTool` tool also **validates the model-generated arguments against its ` See `defineTool`, `validateArgs`, `ToolArgsError`, `SchemaSpec`, `InferArgs`, and `schemaSpecToJsonSchema` in the public API for details. +### Tool-owned UI presentation + +A tool owns how ITS calls render in a UI (an editor's tool-call card, a CLI log line) — a UI plugin must NOT special-case tool names. A `ToolDefinition` may declare two optional, pure, display-only methods: + +- `presentCall(args): ToolCallPresentation | undefined` — the PENDING state: a human-readable `title` (always-visible label), an optional `kind` (`read`/`edit`/`execute`/… for icon/treatment, default `other`), and an optional `rawInput` (the salient input to show in a detail view — e.g. a shell command as a string, NOT the whole args object). +- `presentResult(args, result): ToolResultPresentation | undefined` — the COMPLETED state, given the same `args` and the `{ content, isError }` result: an optional replacement `title` and reformatted `content` (e.g. wrap command output in a fenced ` ```console ` block — a UI-only affordance that must NOT appear in the model-facing `execute` result). + +Returning `undefined` (or omitting a method) tells a UI to fall back to a generic presentation (title = tool name, raw args as input, raw result content). Both methods must be **pure and side-effect-free**: a UI may call them during live streaming AND during a session-log replay, so they depend only on their arguments. With `defineTool`, `args` is the typed `InferArgs` shape; the helper soft-validates before calling (a malformed/older logged arg shape yields `undefined` rather than throwing, since display must never crash a replay). The shapes are provider-neutral — the ACP bridge (`dsh-acp`) maps them to ACP `tool_call`/`tool_call_update` wire fields, and `dsh-tool-bash` is the reference implementation. + +```ts +import { defineTool } from '@deepseek-ai/dsh-tools' + +const bash = defineTool({ + name: 'bash', + description: 'Run a shell command.', + parameters: { + command: { type: 'string', required: true, description: 'The command to run.' }, + description: { type: 'string', required: true, description: 'One-line summary shown in the UI.' }, + }, + async execute(args) { + return [{ type: 'text', text: `ran: ${args.command}` }] + }, + // The model-written description is the readable title; the command is the detail. + presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), + // Wrap the output as a console block for the UI (not in the model-facing result). + presentResult: (_args, result) => { + const block = result.content.length === 1 ? result.content[0] : undefined + if (block === undefined || block.type !== 'text') return undefined + return { content: [{ type: 'text', text: '```console\n' + block.text + '\n```' }] } + }, +}) +``` + ### What is NOT here (TODO) - **Tool shapes review** — when real tools land (e.g. a concurrency-safety hint for parallel execution); phase 1 executes tool calls sequentially. diff --git a/packages/tools/src/index.ts b/packages/tools/src/index.ts index 8e8c106147..9e1e7b74a9 100644 --- a/packages/tools/src/index.ts +++ b/packages/tools/src/index.ts @@ -50,9 +50,87 @@ declare module 'cordis' { // parallel execution — Claude Code partitions read-only tools; phase 1 // executes sequentially). +/** + * Category of a tool call, used by a UI to pick an icon / treatment. A neutral + * vocabulary owned here (NOT an ACP type) so tools describe themselves without + * depending on any client protocol; a UI bridge maps it to its own enum. The + * member set mirrors the common ACP `ToolKind` values; `other` is the default. + */ +export type ToolCallKind = 'read' | 'edit' | 'delete' | 'move' | 'search' | 'execute' | 'fetch' | 'other' + +/** + * How a tool wants ONE of its calls shown in a UI (an editor's tool-call card, + * a CLI log line) BEFORE the result is known — the *pending* state. Provider- + * neutral: a tool returns this from {@link ToolDefinition.presentCall} and a UI + * plugin (e.g. the ACP bridge) maps it to its own wire shape. The tool owns its + * own presentation — the UI must not special-case tool names. + */ +export interface ToolCallPresentation { + /** + * Human-readable, always-visible label describing what THIS call does (e.g. + * the model-written one-line summary of a bash command). Keep it short — a UI + * shows it as a card header / log line. Required: a presentation must have a + * title (a UI falls back to the tool name only when `presentCall` is absent). + */ + title: string + /** Category for icon/treatment; defaults to `other` when omitted. */ + kind?: ToolCallKind + /** + * The salient input to surface in a detail/expanded view — e.g. the bash + * COMMAND itself (as a string), so the title can stay a readable summary + * while the exact command is still visible. Omit to show nothing; a string is + * rendered as-is, an object as pretty JSON. NOT the full raw args object + * unless that is genuinely what a reader wants. + */ + rawInput?: unknown +} + +/** + * How a tool wants the COMPLETED call shown — the *result* state, after + * `execute` returns. Lets the tool reformat its result for a UI distinctly from + * the model-facing text it returned from `execute` (e.g. wrap command output in + * a fenced ```console block for monospace rendering, which the model-facing + * result must NOT carry). All fields optional: a UI keeps the pending-state + * title and renders the raw result content for anything left unset. + */ +export interface ToolResultPresentation { + /** Replacement title for the completed call (e.g. append an exit status). Omit to keep the pending-state title. */ + title?: string + /** + * UI-facing result content (harness {@link ContentBlock}s), reformatted from + * the model-facing result. Omit to let the UI render the raw result content. + * Stays in harness vocabulary; the UI maps these to its own content blocks. + */ + content?: ContentBlock[] +} + /** A registered tool: its schema plus the execution function. */ export interface ToolDefinition extends ToolSchema { execute(args: unknown, exec: ToolExecution): Promise + /** + * Optional: how to present the PENDING state of one call in a UI, derived + * from the call's `args` (parsed arguments, `unknown` — the tool validates/ + * narrows its own input). Returning `undefined` (or omitting the method) tells + * a UI to fall back to a generic presentation (title = tool name, raw args as + * input). Pure and side-effect-free: a UI may call it during live streaming + * AND a session-log replay, so it must depend only on `args`. + */ + presentCall?(args: unknown): ToolCallPresentation | undefined + /** + * Optional: how to present the COMPLETED state, given the same `args` and the + * `result` (`execute`'s content + whether it errored). Returning `undefined` + * (or omitting the method) tells a UI to keep the pending title and render the + * raw result content. Pure and side-effect-free for the same replay reason. + */ + presentResult?(args: unknown, result: ToolResult): ToolResultPresentation | undefined +} + +/** The completed outcome handed to {@link ToolDefinition.presentResult}. */ +export interface ToolResult { + /** The model-facing content `execute` returned (or the error text on failure). */ + content: ContentBlock[] + /** Whether the call failed. */ + isError: boolean } /** One pending tool call, as it flows through the execution waterfall. */ diff --git a/packages/tools/src/schema.ts b/packages/tools/src/schema.ts index 5572e43ff3..5e8887f11b 100644 --- a/packages/tools/src/schema.ts +++ b/packages/tools/src/schema.ts @@ -21,7 +21,7 @@ import type { ContentBlock } from '@deepseek-ai/dsh-llm' import { assertNever, HarnessError } from '@deepseek-ai/dsh-llm' -import type { ToolDefinition, ToolExecution } from './index.ts' +import type { ToolCallPresentation, ToolDefinition, ToolExecution, ToolResult, ToolResultPresentation } from './index.ts' // --------------------------------------------------------------------------- // SchemaSpec — the author-facing per-property type @@ -287,6 +287,22 @@ export interface DefineToolOptions { * casts needed. */ execute(args: InferArgs, exec: ToolExecution): Promise + /** + * Optional: how to present the PENDING state of one call in a UI (an editor + * tool-call card, a CLI log line). `args` is the typed, schema-validated + * argument shape — zero casts. Pure and side-effect-free: a UI may call it + * during live streaming AND a session-log replay, so depend only on `args`. + * The tool owns its presentation so a UI never special-cases tool names. See + * {@link ToolCallPresentation}. + */ + presentCall?(args: InferArgs): ToolCallPresentation | undefined + /** + * Optional: how to present the COMPLETED state, given the typed `args` and the + * `result`. Use it to reformat result content for a UI distinctly from the + * model-facing text (e.g. a fenced ```console block). Pure and side-effect- + * free for the same replay reason. See {@link ToolResultPresentation}. + */ + presentResult?(args: InferArgs, result: ToolResult): ToolResultPresentation | undefined /** Whether the tool requires structured output (default false). */ strict?: boolean } @@ -322,7 +338,11 @@ export function defineTool(options: DefineToolOptions): // Object-literal execute methods don't use `this`; the reference is safe. // eslint-disable-next-line @typescript-eslint/unbound-method const userExecute = options.execute - return { + // eslint-disable-next-line @typescript-eslint/unbound-method + const userPresentCall = options.presentCall + // eslint-disable-next-line @typescript-eslint/unbound-method + const userPresentResult = options.presentResult + const tool: ToolDefinition = { name: options.name, description: options.description, parameters: schemaSpecToJsonSchema(options.parameters) as unknown as Record, @@ -337,4 +357,21 @@ export function defineTool(options: DefineToolOptions): return userExecute(args as InferArgs, exec) }, } + // Presentation is display-only and may run on REPLAY of arbitrary logged args + // (possibly from an older schema), so it must never throw: validate softly and + // fall back to `undefined` (a generic UI presentation) on any mismatch, rather + // than the hard `ToolArgsError` the execute path raises. + if (userPresentCall) { + tool.presentCall = (args: unknown): ToolCallPresentation | undefined => { + if (validateArgs(options.parameters, args).length > 0) return undefined + return userPresentCall(args as InferArgs) + } + } + if (userPresentResult) { + tool.presentResult = (args: unknown, result: ToolResult): ToolResultPresentation | undefined => { + if (validateArgs(options.parameters, args).length > 0) return undefined + return userPresentResult(args as InferArgs, result) + } + } + return tool } diff --git a/packages/tools/tests/tools.spec.ts b/packages/tools/tests/tools.spec.ts index fb42df4492..e42e7b4387 100644 --- a/packages/tools/tests/tools.spec.ts +++ b/packages/tools/tests/tools.spec.ts @@ -814,3 +814,54 @@ describe('defineTool validation (the runtime-validation RFC, part 1)', () => { expect(result.isError).toBe(false) }) }) + +describe('defineTool presentation (presentCall / presentResult)', () => { + it('threads presentCall/presentResult onto the ToolDefinition with typed args', () => { + const tool = defineTool({ + name: 'demo', + description: 'demo', + parameters: { path: { type: 'string', required: true }, n: { type: 'number' } }, + async execute() { return [{ type: 'text', text: 'ok' }] }, + presentCall(args) { + // args is typed { path: string; n?: number } — zero casts. + expectTypeOf(args).toEqualTypeOf<{ path: string; n?: number }>() + return { title: `Open ${args.path}`, kind: 'read', rawInput: args.path } + }, + presentResult(args, result) { + return { title: `Opened ${args.path}`, content: result.content } + }, + }) + expect(tool.presentCall!({ path: '/a', n: 2 })).toEqual({ title: 'Open /a', kind: 'read', rawInput: '/a' }) + expect(tool.presentResult!({ path: '/a' }, { content: [{ type: 'text', text: 'x' }], isError: false })) + .toEqual({ title: 'Opened /a', content: [{ type: 'text', text: 'x' }] }) + }) + + it('a tool without presentCall/presentResult leaves them undefined (UI falls back generically)', () => { + const tool = defineTool({ + name: 'plain', + description: 'plain', + parameters: { x: { type: 'string', required: true } }, + async execute() { return [] }, + }) + expect(typeof tool.presentCall).toBe('undefined') + expect(typeof tool.presentResult).toBe('undefined') + }) + + it('presentCall/presentResult validate softly: malformed args return undefined, never throw (display runs on replay)', () => { + const tool = defineTool({ + name: 'demo', + description: 'demo', + parameters: { path: { type: 'string', required: true } }, + async execute() { return [] }, + presentCall: args => ({ title: args.path }), + presentResult: (args, result) => ({ title: args.path, content: result.content }), + }) + // Unlike execute (which throws ToolArgsError on a mismatch), the display + // methods soft-validate and fall back to undefined so a UI never crashes + // replaying an old/foreign log entry. The ToolDefinition methods take + // `unknown`, so malformed shapes pass without a cast. + expect(tool.presentCall?.({})).toBeUndefined() + expect(tool.presentResult?.({ wrong: 1 }, { content: [], isError: false })).toBeUndefined() + }) +}) + From 8a92338d2f25b2615e9632d4de8af0416e674a58 Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Thu, 18 Jun 2026 10:36:53 +0800 Subject: [PATCH 2/3] fix(acp): address Codex review of the tool-call UI seam MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - schemas() builds the model-facing ToolSchema by EXPLICIT allowlist ({name, description, parameters, strict?}) instead of stripping `execute` — presentCall/presentResult are functions that must never leak into a model request, and an allowlist can't drift when a new ToolDefinition member lands. - session/load replay uses a THROWAWAY ToolPresenter, not record.presenter, so a historical interrupted-mid-tool turn (tool/call with no tool/result) can't leave stale in-flight state on the live presenter that serves later events. - ToolPresenter.call/result contain a throwing presentCall/presentResult: log via an onError sink and fall back to the generic presentation, so a buggy display callback can never fail a live turn or a load replay. - acp README inject list now includes `tools`. - remove a stray blank line at EOF (git diff --check gate). Regressions added: schemas() drops presenter callbacks (+ keeps `strict`); session/load replays a tool call with the tool-owned presentation; a throwing presenter is contained (direct + through the real bridge) with and without an onError sink. --- packages/acp/README.md | 2 +- packages/acp/src/index.ts | 60 +++++++++++++++++----- packages/acp/tests/load.spec.ts | 64 +++++++++++++++++++++++- packages/acp/tests/stream-update.spec.ts | 52 +++++++++++++++++++ packages/acp/tests/turns.spec.ts | 27 ++++++++++ packages/tools/src/index.ts | 20 +++++--- packages/tools/tests/tools.spec.ts | 34 ++++++++++++- 7 files changed, 237 insertions(+), 22 deletions(-) diff --git a/packages/acp/README.md b/packages/acp/README.md index f515ae8293..9818c7b13a 100644 --- a/packages/acp/README.md +++ b/packages/acp/README.md @@ -8,7 +8,7 @@ It is a **client-driver / UI plugin**, the structured analogue of the readline ` `apply(ctx, config)` — wires an `AgentSideConnection` (from `@agentclientprotocol/sdk`) to `process.stdin`/`process.stdout` and implements the ACP `Agent` method surface. -`inject: ['agents', 'sessions', 'sessionPersistence']` — programs against the interface packages only (never `dsh-agent-loop`). `sessionPersistence` is required because `initialize` advertises `loadSession: true`. +`inject: ['agents', 'sessions', 'sessionPersistence', 'tools']` — programs against the interface packages only (never `dsh-agent-loop`). `sessionPersistence` is required because `initialize` advertises `loadSession: true`; `tools` lets a tool own how its calls render (`presentCall`/`presentResult`) — the bridge looks the definition up by name and falls back to a generic presentation when a tool declares none (see Tool-call presentation). ### Config diff --git a/packages/acp/src/index.ts b/packages/acp/src/index.ts index 793685072b..283b8fc151 100644 --- a/packages/acp/src/index.ts +++ b/packages/acp/src/index.ts @@ -60,7 +60,7 @@ import { import type { ContentBlock } from '@deepseek-ai/dsh-llm' import type { Agent, AgentStatus } from '@deepseek-ai/dsh-agent' import type { SessionEvent } from '@deepseek-ai/dsh-session' -import type { ToolCallKind, ToolRegistry } from '@deepseek-ai/dsh-tools' +import type { ToolCallKind, ToolCallPresentation, ToolRegistry, ToolResultPresentation } from '@deepseek-ai/dsh-tools' // Side-effect type import: declaration-merges `ctx.sessionPersistence` onto // Context (the bridge injects it and reads `list()` for load cwd validation). import type {} from '@deepseek-ai/dsh-session-persistence' @@ -193,6 +193,9 @@ export function apply(ctx: Context, config: AcpConfig): void { const sessionPersistence = ctx.sessionPersistence const logger = ctx.logger const tools = ctx.tools + // A new ToolPresenter per session (and a throwaway per load replay), each given + // this warn sink so a throwing tool presenter is logged, not propagated. + const makePresenter = (): ToolPresenter => new ToolPresenter(tools, (message) => { logger.warn(message) }) // Live sessions keyed by id (RFC 011 multi-session), plus an agent→sessionId // reverse map so `agent/*` events (which carry only the Agent) demux in O(1). @@ -406,7 +409,7 @@ export function apply(ctx: Context, config: AcpConfig): void { agentOptions: agentOptions(config), }) bySession.set(agent, sessionId) - sessions.set(sessionId, { sessionId, agent, presenter: new ToolPresenter(tools), inflight: undefined }) + sessions.set(sessionId, { sessionId, agent, presenter: makePresenter(), inflight: undefined }) return Promise.resolve({ sessionId }) }, @@ -459,18 +462,24 @@ export function apply(ctx: Context, config: AcpConfig): void { throw invalidParams('connection closed during session/load') } bySession.set(agent, params.sessionId) - const record: SessionRecord = { sessionId: params.sessionId, agent, presenter: new ToolPresenter(tools), inflight: undefined } + const record: SessionRecord = { sessionId: params.sessionId, agent, presenter: makePresenter(), inflight: undefined } sessions.set(params.sessionId, record) // Replay the persisted event log to the client as session/update. Use // the raw event log (NOT deriveMessages, which drops assistant/chunk // and trace events): RFC 010's load contract reconstructs the streamed // turns — user prompts (user/message → user_message_chunk), assistant - // text and reasoning (assistant/chunk), and tool calls/results. The - // record's presenter pairs each tool/call with its tool/result as the - // log replays in order, so the replayed tool cards render identically - // to the live ones. + // text and reasoning (assistant/chunk), and tool calls/results. + // + // Replay through a THROWAWAY presenter, NOT `record.presenter`: a + // historical turn that was interrupted mid-tool (a `tool/call` with no + // matching `tool/result` in the persisted log) would otherwise leave a + // stale in-flight entry on the live presenter, which then serves all + // future live events for this session. The throwaway pairs call→result + // as the log replays in order (same as live) and is discarded after, + // so the record's presenter starts clean for the post-load live stream. + const replayPresenter = makePresenter() for (const event of agent.session.events) { - streamSessionEventUpdate(params.sessionId, event, notify, record.presenter) + streamSessionEventUpdate(params.sessionId, event, notify, replayPresenter) } return {} } finally { @@ -785,13 +794,31 @@ interface ResolvedResultPresentation { export class ToolPresenter { private readonly pending = new Map() - constructor(private readonly tools: Pick) {} + /** + * @param tools the registry to resolve tool definitions by name. + * @param onError invoked when a tool's `presentCall`/`presentResult` THROWS; + * the presenter swallows the error and falls back to the generic + * presentation so a buggy display callback can never fail a live turn or a + * `session/load` replay (AGENTS.md "contain callback exceptions at the + * boundary"). Defaults to a no-op for callers that don't supply a logger. + */ + constructor( + private readonly tools: Pick, + private readonly onError: (message: string) => void = () => {}, + ) {} /** Pending-state presentation for a `tool/call`; remembers `(name, args)` for the matching result. */ call(callId: string, name: string, argsJson: string): ResolvedCallPresentation { const args = parseToolArguments(argsJson) this.pending.set(callId, { name, args }) - const present = this.tools.get(name)?.presentCall?.(args) + let present: ToolCallPresentation | undefined + try { + present = this.tools.get(name)?.presentCall?.(args) + } catch (error: unknown) { + // A throwing presentCall must not break streaming: log and fall back. + this.onError(`acp: tool "${name}" presentCall threw, using generic presentation: ${String(error)}`) + present = undefined + } if (present === undefined) { // No tool-owned presentation: fall back to the tool name as the title and // the full parsed args as the raw input (the pre-seam behavior). @@ -804,9 +831,16 @@ export class ToolPresenter { result(callId: string, content: ContentBlock[], isError: boolean): ResolvedResultPresentation { const call = this.pending.get(callId) this.pending.delete(callId) - const present = call !== undefined - ? this.tools.get(call.name)?.presentResult?.(call.args, { content, isError }) - : undefined + // No remembered call (unknown/late callId) → nothing to present from; raw content. + if (call === undefined) return { content } + let present: ToolResultPresentation | undefined + try { + present = this.tools.get(call.name)?.presentResult?.(call.args, { content, isError }) + } catch (error: unknown) { + // A throwing presentResult must not break streaming/replay: log + fall back. + this.onError(`acp: tool "${call.name}" presentResult threw, using raw result: ${String(error)}`) + present = undefined + } if (present === undefined) return { content } return { content: present.content ?? content, diff --git a/packages/acp/tests/load.spec.ts b/packages/acp/tests/load.spec.ts index 3bed1f836d..3ffef1a87e 100644 --- a/packages/acp/tests/load.spec.ts +++ b/packages/acp/tests/load.spec.ts @@ -4,7 +4,8 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { PROTOCOL_VERSION } from '@agentclientprotocol/sdk' import { SessionId } from '@deepseek-ai/dsh-session' -import { makeBridgeHarness, textResponse, type BridgeHarness, type CapturedUpdate } from './harness.ts' +import { defineTool } from '@deepseek-ai/dsh-tools' +import { makeBridgeHarness, textResponse, toolCallResponse, type BridgeHarness, type CapturedUpdate } from './harness.ts' /** Concatenate the text of all agent_message_chunk updates. */ function messageText(updates: CapturedUpdate[]): string { @@ -56,6 +57,67 @@ describe('acp bridge — session/load replay', () => { expect(userText).toBe('remember this') }) + it('replays a persisted tool call with the TOOL-OWNED presentation (title/rawInput/console output)', async () => { + // A turn with a tool call is persisted, then loaded by a fresh bridge. The + // replayed tool_call/tool_call_update must carry the tool's OWN presentation + // (presentCall/presentResult) — identical to how they streamed live — using + // a throwaway presenter that pairs call→result as the log replays in order. + live = await makeBridgeHarness({ + storageDir, + script: [toolCallResponse('c1', 'bash', { command: 'ls -la', description: 'List files' }), textResponse('done')], + }) + live.ctx.tools.register(defineTool({ + name: 'bash', + description: 'run a command', + parameters: { + command: { type: 'string', required: true }, + description: { type: 'string', required: true }, + }, + async execute() { return [{ type: 'text', text: 'a.txt\nb.txt\n' }] }, + presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), + presentResult: (_args, result) => { + const block = result.content.length === 1 ? result.content[0] : undefined + if (block === undefined || block.type !== 'text') return undefined + return { content: [{ type: 'text', text: `\`\`\`console\n${block.text.trimEnd()}\n\`\`\`` }] } + }, + })) + await live.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: {} }) + const { sessionId } = await live.client.newSession({ cwd: process.cwd(), mcpServers: [] }) + await live.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'list' }] }) + await live.dispose() + live = undefined + + // A fresh bridge — which must ALSO have the tool registered, since the + // presentation is resolved from the live registry at replay time — loads it. + loader = await makeBridgeHarness({ storageDir, script: [] }) + loader.ctx.tools.register(defineTool({ + name: 'bash', + description: 'run a command', + parameters: { + command: { type: 'string', required: true }, + description: { type: 'string', required: true }, + }, + async execute() { return [] }, + presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), + presentResult: (_args, result) => { + const block = result.content.length === 1 ? result.content[0] : undefined + if (block === undefined || block.type !== 'text') return undefined + return { content: [{ type: 'text', text: `\`\`\`console\n${block.text.trimEnd()}\n\`\`\`` }] } + }, + })) + await loader.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: {} }) + await loader.client.loadSession({ sessionId, cwd: process.cwd(), mcpServers: [] }) + + const call = loader.updates.find(u => u.sessionUpdate === 'tool_call') + expect(call).toMatchObject({ toolCallId: 'c1', title: 'List files', kind: 'execute', rawInput: 'ls -la' }) + const update = loader.updates.find(u => u.sessionUpdate === 'tool_call_update') + expect(update).toMatchObject({ + toolCallId: 'c1', + status: 'completed', + content: [{ type: 'content', content: { type: 'text', text: '```console\na.txt\nb.txt\n```' } }], + }) + }) + it('a load whose resume finishes after a client disconnect leaks no live session', async () => { // A session/load is mid-resume() when the client transport closes. The load // must NOT end up with a live registered agent for the connection that is diff --git a/packages/acp/tests/stream-update.spec.ts b/packages/acp/tests/stream-update.spec.ts index 4328b5fb6b..8a22bc71f1 100644 --- a/packages/acp/tests/stream-update.spec.ts +++ b/packages/acp/tests/stream-update.spec.ts @@ -227,6 +227,58 @@ describe('ToolPresenter (tool-owned presentation via the tool registry)', () => })) expect(late).toMatchObject({ content: [{ type: 'content', content: { type: 'text', text: 'late' } }] }) }) + + it('a THROWING presentCall/presentResult is contained: generic fallback + onError, never propagates', () => { + // A buggy tool whose display callbacks throw must NOT fail a live turn or a + // session/load replay (AGENTS.md "contain callback exceptions at the + // boundary"). The presenter swallows the throw, reports via onError, and + // falls back to the generic presentation. + const boom: ToolDefinition = { + name: 'boom', + description: 'b', + parameters: {}, + execute: async () => [], + presentCall: () => { throw new Error('call boom') }, + presentResult: () => { throw new Error('result boom') }, + } + const errors: string[] = [] + const presenter = new ToolPresenter(registryOf(boom), msg => errors.push(msg)) + const updates = updatesWith( + presenter, + evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'boom', arguments: '{"a":1}' }), + evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'raw' }], isError: false }), + ) + // tool/call fell back to title=name, raw args as rawInput. + expect(updates[0]).toMatchObject({ sessionUpdate: 'tool_call', title: 'boom', kind: 'other', rawInput: { a: 1 } }) + // tool/result fell back to the raw content. + expect(updates[1]).toMatchObject({ sessionUpdate: 'tool_call_update', content: [{ type: 'content', content: { type: 'text', text: 'raw' } }] }) + // Both throws were reported, not propagated. + expect(errors).toHaveLength(2) + expect(errors[0]).toContain('presentCall threw') + expect(errors[1]).toContain('presentResult threw') + }) + + it('contains a throwing presenter even with the DEFAULT (no-op) onError sink', () => { + // Constructed without an onError sink (the default `() => {}`): a throwing + // presenter is still swallowed and falls back generically — the absence of a + // logger must not turn a display bug into a propagated exception. + const boom: ToolDefinition = { + name: 'boom', + description: 'b', + parameters: {}, + execute: async () => [], + presentCall: () => { throw new Error('call boom') }, + presentResult: () => { throw new Error('result boom') }, + } + const presenter = new ToolPresenter(registryOf(boom)) + const updates = updatesWith( + presenter, + evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'boom', arguments: '{}' }), + evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'raw' }], isError: false }), + ) + expect(updates[0]).toMatchObject({ sessionUpdate: 'tool_call', title: 'boom' }) + expect(updates[1]).toMatchObject({ sessionUpdate: 'tool_call_update', content: [{ type: 'content', content: { type: 'text', text: 'raw' } }] }) + }) }) describe('agentOptions', () => { diff --git a/packages/acp/tests/turns.spec.ts b/packages/acp/tests/turns.spec.ts index 7258c590b4..101f6849fe 100644 --- a/packages/acp/tests/turns.spec.ts +++ b/packages/acp/tests/turns.spec.ts @@ -111,6 +111,33 @@ describe('acp bridge — turn outcomes', () => { }) }) + it('a throwing tool presenter does not break the turn: the bridge falls back generically', async () => { + // A buggy tool whose presentCall throws must not fail the live turn — the + // bridge's presenter contains the throw (logging via its onError sink) and + // falls back to the generic title=name presentation. Exercises the real + // bridge wiring of the per-session presenter's error sink. + harness = await makeBridgeHarness({ + storageDir, + script: [toolCallResponse('c1', 'kaboom', { x: 1 }), textResponse('done')], + }) + harness.ctx.tools.register(defineTool({ + name: 'kaboom', + description: 'explodes when presented', + parameters: { x: { type: 'number' } }, + async execute() { return [{ type: 'text', text: 'ok' }] }, + presentCall: () => { throw new Error('present boom') }, + })) + const sessionId = await newSession(harness) + const res = await harness.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'go' }] }) + expect(res.stopReason).toBe('end_turn') // the turn completed despite the throw + + const call = harness.updates.find(u => u.sessionUpdate === 'tool_call') + // Generic fallback: title is the tool name, raw args as rawInput. + expect(call).toMatchObject({ toolCallId: 'c1', title: 'kaboom', kind: 'other', rawInput: { x: 1 } }) + const update = harness.updates.find(u => u.sessionUpdate === 'tool_call_update') + expect(update).toMatchObject({ toolCallId: 'c1', status: 'completed' }) + }) + it('a failing tool yields a failed tool_call_update', async () => { harness = await makeBridgeHarness({ storageDir, diff --git a/packages/tools/src/index.ts b/packages/tools/src/index.ts index 9e1e7b74a9..c967f0fba2 100644 --- a/packages/tools/src/index.ts +++ b/packages/tools/src/index.ts @@ -244,14 +244,22 @@ export class ToolRegistry extends Service { } /** - * Return all registered tool schemas, stripped of their `execute` functions. - * These are exactly what gets sent to the model via the system-prompt - * assembly. + * Return all registered tool schemas — exactly the model-facing fields + * (`name`, `description`, `parameters`, and `strict` when set), as sent to the + * model via the system-prompt assembly. Constructed EXPLICITLY rather than by + * stripping known non-schema members: a `ToolDefinition` also carries + * `execute` and the optional `presentCall`/`presentResult` UI callbacks, and + * those (especially the functions) must never leak into a model request. An + * allowlist can't drift when a new non-schema member is added to the + * definition; a denylist (rest-destructure) would silently leak it. */ schemas(): ToolSchema[] { - // Rest-destructure to drop `execute`; the unused binding is the idiom. - // eslint-disable-next-line @typescript-eslint/unbound-method, @typescript-eslint/no-unused-vars - return [...this.store.values()].map(({ execute, ...schema }) => schema) + return [...this.store.values()].map(({ name, description, parameters, strict }): ToolSchema => ({ + name, + description, + parameters, + ...strict !== undefined ? { strict } : {}, + })) } /** diff --git a/packages/tools/tests/tools.spec.ts b/packages/tools/tests/tools.spec.ts index e42e7b4387..9706790d55 100644 --- a/packages/tools/tests/tools.spec.ts +++ b/packages/tools/tests/tools.spec.ts @@ -41,6 +41,39 @@ describe('ToolRegistry', () => { expect(assembly.tools.map(t => t.name)).toEqual(['echo']) }) + it('schemas() drops the UI presentation callbacks — they must never reach the model', async () => { + const ctx = await setup() + // A tool that declares presentCall/presentResult (functions). schemas() feeds + // the system-prompt assembly → the model request, so those callbacks (and + // `execute`) must be stripped: a function in the JSON tool schema would + // corrupt the request. schemas() is an explicit allowlist, so it can't leak. + ctx.tools.register(defineTool({ + name: 'present', + description: 'has presenters', + parameters: { x: { type: 'string', required: true } }, + async execute() { return [] }, + presentCall: args => ({ title: args.x }), + presentResult: (args, result) => ({ title: args.x, content: result.content }), + })) + const schema = ctx.tools.schemas()[0] as unknown as Record + expect(Object.keys(schema).sort()).toEqual(['description', 'name', 'parameters']) + expect(schema.presentCall).toBeUndefined() + expect(schema.presentResult).toBeUndefined() + expect(schema.execute).toBeUndefined() + }) + + it('schemas() preserves `strict` when set (allowlist keeps the model-facing fields)', async () => { + const ctx = await setup() + ctx.tools.register(defineTool({ + name: 'strict-tool', + description: 'd', + parameters: { x: { type: 'string', required: true } }, + strict: true, + async execute() { return [] }, + })) + expect(ctx.tools.schemas()[0]).toMatchObject({ name: 'strict-tool', strict: true }) + }) + it('executes a tool and returns its content', async () => { const ctx = await setup() ctx.tools.register(echoTool) @@ -864,4 +897,3 @@ describe('defineTool presentation (presentCall / presentResult)', () => { expect(tool.presentResult?.({ wrong: 1 }, { content: [], isError: false })).toBeUndefined() }) }) - From 8acafe918fed6bed8afd94287e0ba7dbeb89453c Mon Sep 17 00:00:00 2001 From: Tianyi Cui <53024+tianyicui@users.noreply.github.com> Date: Thu, 18 Jun 2026 11:23:12 +0800 Subject: [PATCH 3/3] feat(acp): show the command in execute titles; test via the real bash tool; RFC for terminal rendering MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - bash presentCall title is now "description — command" (e.g. "List files in src — ls -la src"). An execute-kind ACP card HIDES rawInput (Zed renders it only for non-terminal tools), so the command must ride in the always-visible title to be seen — matching how claude-agent-acp/codex-acp title execute tools. The command stays in rawInput too for non-execute UIs that show it. - Rework the acp tool-call presentation tests (turns + load replay) to drive the REAL dsh-tool-bash + dsh-bash-local via a new makeBridgeHarness({ withBash }) option, running an actual `echo` — instead of an inline fake bash tool. The mock MODEL still scripts the call (deterministic, no key), but the tool and executor are real, so the test verifies the shipping presentCall/presentResult. - AGENTS.md: add the principle "prefer the REAL implementation over a mock/ stand-in in tests" (mock only the expensive/non-deterministic boundary). - RFC (proposed): the ACP terminal sub-protocol + command classification — the capability-gated rich rendering (live cwd-header terminal card, classify a `cat` as a read / `grep` as a search) that the reference adapters do; the fenced ```console text block stays the no-capability baseline. Studied codex-acp, claude-agent-acp, and Zed's renderer to ground it. --- AGENTS.md | 1 + docs/rfc/README.md | 1 + ...6-06-18-acp-terminal-and-tool-rendering.md | 49 ++++++++++++++ packages/acp/README.md | 4 +- packages/acp/package.json | 2 + packages/acp/src/index.ts | 11 +++- packages/acp/tests/harness.ts | 14 ++++ packages/acp/tests/load.spec.ts | 64 ++++++------------- packages/acp/tests/turns.spec.ts | 53 +++++++-------- packages/tool-bash/README.md | 2 +- packages/tool-bash/src/index.ts | 17 +++-- packages/tool-bash/tests/tools.spec.ts | 6 +- pnpm-lock.yaml | 6 ++ 13 files changed, 144 insertions(+), 86 deletions(-) create mode 100644 docs/rfc/proposed/2026-06-18-acp-terminal-and-tool-rendering.md diff --git a/AGENTS.md b/AGENTS.md index c1f183ee04..d4b8474865 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -121,6 +121,7 @@ Dev/test/demo run **unbuilt** via tsx + the `paths` map in the root `tsconfig.js - **Merging PRs**: always merge with a **merge commit** (`gh pr merge --merge`), never squash or rebase. The per-PR commit history is intentional — review-fix commits, regression-test commits, and the reasoning in each message are part of the record — and squashing flattens it away. - **TODO markers**: use `FIXME`/`TODO`/`XXX` to flag known issues by urgency — see [docs/development.md](docs/development.md) for the semantics of each. - **Tests**: vitest, colocated under `packages//tests/*.spec.ts`. Every registry needs an HMR-safety test (dispose the contributing fiber, assert cleanup). **Excessive tests are welcome** — when in doubt, write the test; err on the side of covering edge cases, error paths, event ordering, and concurrency races even if they seem unlikely. Review findings get regression tests (see `packages/agent-loop/tests/review-fixes.spec.ts`). The same generosity applies to **real-API (with-key) e2e tests — inference is cheap here (we are DeepSeek), so do not ration them**: cover the agent's real flows (a real prompt that writes a file, multi-turn, tool use, cancellation) and run them frequently while developing, especially cheap **smoke tests** that boot the real example and check the world. A green mock/no-key suite proves the plumbing, not the product — the with-key smoke test is what catches "green units, broken product". See § Secrets / .env for the with-key policy and why self-skip is a CI accommodation, not a verdict that real-API tests are expensive. +- **Prefer the REAL implementation over a mock/stand-in in tests.** When the genuine collaborator is available in the repo, wire it up instead of hand-rolling a fake — a test that registers an inline `defineTool({ name: 'bash', … })` to stand in for `dsh-tool-bash` proves the *bridge* moves bytes but not that the *shipping tool* renders the way the test asserts; the two drift and the test passes while the product is wrong. Mock only the genuinely expensive/non-deterministic boundary (the LLM adapter, the network, the clock) and keep everything downstream real: a bridge tool-call test runs the scripted mock MODEL but the REAL tool + REAL executor (e.g. `makeBridgeHarness({ withBash: true })` plugs `dsh-bash-local` + `dsh-tool-bash` and runs an actual `echo`), so it verifies the actual `presentCall`/`presentResult` an editor sees. This is the unit-test echo of "verify the world, not a synthetic stand-in" (see § Defensive patterns) — a fake you wrote will agree with whatever you assumed; the real thing won't. ## Defensive patterns (hard-won) diff --git a/docs/rfc/README.md b/docs/rfc/README.md index 6a4ddb6f05..90fd5fe6d4 100644 --- a/docs/rfc/README.md +++ b/docs/rfc/README.md @@ -31,6 +31,7 @@ Do NOT write one for a mechanical or local choice (a variable name, a one-file r | [Multiplex concurrent ACP sessions over one connection](proposed/2026-06-14-acp-multi-session.md) | 2026-06-14 | | [Optional Code Mode — model writes TypeScript against an SDK of all tools](proposed/2026-06-15-optional-code-mode.md) | 2026-06-15 | | [Runtime schemas for the event vocabulary (Zod vs the merge-extensible-map pattern)](proposed/2026-06-16-typed-event-schemas.md) | 2026-06-16 | +| [Rich ACP bash rendering — the terminal sub-protocol and command classification](proposed/2026-06-18-acp-terminal-and-tool-rendering.md) | 2026-06-18 | ## Implemented diff --git a/docs/rfc/proposed/2026-06-18-acp-terminal-and-tool-rendering.md b/docs/rfc/proposed/2026-06-18-acp-terminal-and-tool-rendering.md new file mode 100644 index 0000000000..e138609db0 --- /dev/null +++ b/docs/rfc/proposed/2026-06-18-acp-terminal-and-tool-rendering.md @@ -0,0 +1,49 @@ +# RFC: Rich ACP bash rendering — the terminal sub-protocol and command classification + +Status: proposed + +## Problem + +The ACP bridge now lets each tool own its call rendering via `presentCall`/`presentResult` (see [tool-call UI presentation](2026-06-14-acp-agent-client-protocol.md) and `packages/tools`). For `bash` we surface the model's `description` plus the command as the `tool_call` title, `kind: 'execute'`, and the completed output wrapped in a fenced ` ```console ` text block. + +That is a correct, capability-free MVP, but it is not how the reference editors render a *terminal* tool at its best. Two gaps: + +1. **No live terminal card.** An editor like Zed has a dedicated terminal tool-call card — a header showing the working directory, the command, a copy button, and **streaming** output with an exit-status pill — but it only uses that card when the `tool_call`'s `content` is an ACP `terminal` block (`{ type: 'terminal', terminalId }`), not a text block. With a text block the command output appears only as static markdown *after the turn completes*; there is no live stream and no cwd header. (Zed also HIDES `rawInput` for `kind: 'execute'`, which is why our command currently has to ride inside the title.) + +2. **No command classification.** A bash invocation is opaque — `bash -lc "sed -n 1,40p foo.ts"` is really a file read, `rg foo` is a search. The reference adapters classify common commands and present them with a *semantic* kind/title/locations (a `read` card titled "Read file 'foo.ts'" with a follow-along file location, a `search` card), falling back to a terminal card only for an unrecognized command. This is what makes one bash call render with a search icon and "List …" while the next renders as a raw terminal. + +## What the reference adapters do (studied 2026-06-18) + +- **`codex-acp`** (`CodexToolCallMapper.ts`): classifies each command into `commandActions`. A recognized action maps to a semantic update — `read` → `{kind:'read', title:"Read file '…'", locations:[{path}]}`, `search` → `{kind:'search', title:"Search for '…' in …"}`, `listFiles` → `{kind:'read', title:"List files in '…'"}`. An `unknown` action becomes a terminal card: `{kind:'execute', title: stripShellPrefix(command), content:[{type:'terminal', terminalId}], _meta:{terminal_info:{cwd, terminal_id}}}`. The `_meta.terminal_info.cwd` is what renders the working directory as the card header. +- **`claude-agent-acp`** (`tools.ts`): gates on `clientCapabilities._meta.terminal_output`. WITH it: a terminal content block plus `_meta.terminal_{info,output,exit}` (output + exit code). WITHOUT it: the same fenced ` ```console ` text-block fallback this bridge ships today. Title is the command; the model's `description` (when present) is shown as content. +- **Zed** (`crates/agent_ui/.../thread_view.rs`, `crates/acp_thread/.../acp_thread.rs`): `render_terminal_tool_call` reads the terminal's `working_dir` as the header and `tool_call.label` (the title) as the command; a non-terminal text `content` block renders via `render_markdown_output`. `should_show_raw_input = !is_terminal_tool` confirms `rawInput` is suppressed for execute-kind cards. + +The full terminal experience is an ACP **sub-protocol**, not just a content shape: the client advertises a terminal capability, and the agent drives `terminal/create` → streams via `terminal/output` → `terminal/release`, attaching the `terminalId` to the `tool_call` content. That is a cross-seam feature (bridge ⇄ `dsh-bash` executor ⇄ client), which is why it is deferred to this RFC rather than folded into the presentation-seam PR. + +## Proposal + +Two independent, separately shippable pieces. Both build on the existing tool-owned presentation seam — neither reintroduces tool-name special-casing in the bridge. + +### A. Terminal content type + cwd metadata (capability-gated) + +1. In `initialize`, read the client's terminal capability (`clientCapabilities.terminal` / the `_meta.terminal_output` convention the references use) and remember it per connection. +2. Extend the `dsh-tools` presentation vocabulary so a tool can ask for a terminal rendering — e.g. a `ToolResultPresentation`/`ToolCallPresentation` variant carrying `{ kind: 'terminal', cwd, terminalId? }` (provider-neutral; the bridge maps it to the ACP `terminal` content block + `_meta.terminal_info`). `dsh-tool-bash` returns it for `bash` when a cwd is known. +3. When the client supports it, the bridge maps that to `content:[{type:'terminal', terminalId}]` + `_meta.terminal_info.{cwd,terminal_id}`; otherwise it keeps the current ` ```console ` text fallback. The fenced-text path stays the guaranteed baseline. +4. *(Stretch)* drive live streaming through the real `terminal/*` methods so output appears as it is produced, with an exit-status pill — this needs a streaming seam on `dsh-bash` (the executor already has the process; it would push incremental output to the bridge). Scope this as a follow-up sub-step; steps 1–3 already give the cwd-header card with output attached at completion. + +### B. Command classification (capability-free) + +A small, pure classifier (in `dsh-tool-bash`, since it owns the bash schema) maps a command string to an optional semantic presentation: detect common read/search/list shapes (`cat`/`sed -n`/`head`/`tail` → `read` + a `path` location; `grep`/`rg` → `search`; `ls` → list) and return the richer `ToolCallPresentation` (`kind`, a human title, `locations`). Anything unrecognized falls through to the current execute/terminal presentation. This needs a `locations?: ToolCallLocation[]`-style field on `ToolCallPresentation` (neutral `{ path, line? }`), which the bridge maps to ACP `tool_call.locations` to drive editor "follow-along". + +Classification is best-effort and explicitly fallible: a misparse must degrade to the plain terminal card, never mislabel destructively (e.g. never title a `rm` as a "read"). Keep the matcher conservative and unit-test each recognized shape plus the fallthrough. + +## Risks / trade-offs + +- **Terminal sub-protocol is cross-seam and stateful.** Live streaming couples the bridge, the `dsh-bash` executor, and the client's terminal lifecycle; getting disposal/cancel right (release the terminal on turn end, abort, and disconnect) is the hard part — it must honor the same quiescence rules as the rest of the bridge. Steps A1–A3 (static cwd header + output at completion) are low-risk; A4 (live streaming) is where the lifecycle complexity lives. +- **Capability detection must stay honest.** Advertise/emit terminal content only when the client opted in; the text fallback is the contract for everyone else, so it must never regress. +- **Classification can mislead.** A wrong guess is worse than no guess. Bias to the terminal fallback; treat the classifier as additive polish, not a correctness path. (Security note: classification is display-only — it must never change what actually executes.) +- **Provider-neutral vocabulary creep.** Adding `terminal`/`locations` to `ToolCallPresentation` widens the `dsh-tools` surface. Keep the additions neutral (no ACP types leak into `dsh-tools`) and only as rich as a second consumer would also want. + +## Out of scope / non-goals + +The MVP shipped in the tool-call-UI PR (description+command title, `kind:'execute'`, ` ```console ` output fallback) stays the baseline and the no-capability default. This RFC is purely additive polish on top of it. diff --git a/packages/acp/README.md b/packages/acp/README.md index 9818c7b13a..3cc54dff3b 100644 --- a/packages/acp/README.md +++ b/packages/acp/README.md @@ -42,10 +42,12 @@ Each session runs in its own workspace, recorded as the session's `SessionHeader ## Tool-call presentation -How a tool call renders in the editor is owned by the TOOL, not the bridge — the bridge never special-cases tool names. Each tool may declare `presentCall(args)` (pending state: a human-readable `title`, a `kind` for the icon, and the salient `rawInput` to show in a detail view) and `presentResult(args, result)` (completed state: an optional replacement `title` and reformatted `content`) on its `dsh-tools` definition. The bridge looks the definition up by name in `ctx.tools` and maps the neutral `ToolCallPresentation`/`ToolResultPresentation` to the ACP `tool_call`/`tool_call_update` wire shapes. A tool that declares neither gets a generic fallback (title = tool name, raw parsed args as `rawInput`, kind inferred from the name). For example `dsh-tool-bash` makes the model-written one-line `description` the title ("List files in the current directory"), the exact `command` the `rawInput`, `kind: 'execute'`, and wraps the completed output in a fenced ` ```console ` block. +How a tool call renders in the editor is owned by the TOOL, not the bridge — the bridge never special-cases tool names. Each tool may declare `presentCall(args)` (pending state: a human-readable `title`, a `kind` for the icon, and the salient `rawInput` to show in a detail view) and `presentResult(args, result)` (completed state: an optional replacement `title` and reformatted `content`) on its `dsh-tools` definition. The bridge looks the definition up by name in `ctx.tools` and maps the neutral `ToolCallPresentation`/`ToolResultPresentation` to the ACP `tool_call`/`tool_call_update` wire shapes. A tool that declares neither gets a generic fallback (title = tool name, raw parsed args as `rawInput`, kind inferred from the name). For example `dsh-tool-bash` sets the title to the model `description` + the exact `command` ("List files in src — ls -la src"), `kind: 'execute'`, the `command` as `rawInput`, and wraps the completed output in a fenced ` ```console ` block. (The command goes in the title because an editor hides `rawInput` for execute-kind cards — Zed renders it only for non-terminal tools.) The `tool/result` session event carries only `{ callId, content, isError }` — not the tool name or args — so to call a tool's `presentResult` the bridge keeps a small per-session map from `callId` to the in-flight call's `(name, args)`, populated on `tool/call` and removed as each result is presented (it holds only currently-in-flight calls, never finished ones). This is bridge-local state — NOT a change to the event schema or a core service. The map lives on the `SessionRecord`, so two concurrent sessions never cross their in-flight tool state; a `session/load` replay uses a throwaway presenter that pairs each `tool/call` with its `tool/result` as the log replays in order, so replayed tool cards render identically to live ones. +A richer rendering — the ACP **terminal** content type (a live cwd-header terminal card with streaming output) and command classification (a `cat` shown as a `read`, a `grep` as a `search`) — is a capability-gated follow-up; the ` ```console ` text block here is the guaranteed baseline for clients without the terminal capability. See [the terminal-rendering RFC](../../docs/rfc/proposed/2026-06-18-acp-terminal-and-tool-rendering.md). + ## Settle-exactly-once A `session/prompt` resolves (or rejects) exactly once, keyed off the canonical session log (the `session/event` stream), NOT the `agent/turn-start`/`agent/turn-end` events. One listener captures the prompt's owning turn from the log's `turn/start` and settles on the matching `turn/end` — the one signal that always fires (`closeTurn` appends it unconditionally, even when a boundary emit throws and the `agent/turn-end` EVENT is skipped). A prompt settles only on ITS OWN turn (`inflight.turn === turn/end.turn`), so a stale `turn/end` for a previously-cancelled turn whose end arrives late can never settle the wrong prompt. A turn that ends `error` REJECTS the RPC with an internal error carrying the failure message (ACP has no error stop reason); every other reason resolves via the codec. As a fallback, when the agent settles to `idle`/`disposed` with a prompt still pending — e.g. a `session/event` listener registered before the bridge threw and starved the bridge's listener — an `agent/status` handler reconciles the prompt from the log (the owning turn's `turn/end`, or `cancelled` if the turn was torn down without one). An empty/whitespace prompt is rejected up front — it would queue no work, so no turn would start and the RPC would hang. diff --git a/packages/acp/package.json b/packages/acp/package.json index 9cac888518..0a4a890658 100644 --- a/packages/acp/package.json +++ b/packages/acp/package.json @@ -35,11 +35,13 @@ "devDependencies": { "@deepseek-ai/dsh-agent": "workspace:^", "@deepseek-ai/dsh-agent-loop": "workspace:^", + "@deepseek-ai/dsh-bash-local": "workspace:^", "@deepseek-ai/dsh-llm": "workspace:^", "@deepseek-ai/dsh-session": "workspace:^", "@deepseek-ai/dsh-session-persistence": "workspace:^", "@deepseek-ai/dsh-session-persistence-jsonl": "workspace:^", "@deepseek-ai/dsh-system-prompt": "workspace:^", + "@deepseek-ai/dsh-tool-bash": "workspace:^", "@deepseek-ai/dsh-tools": "workspace:^", "cordis": "^4.0.0-rc.6" } diff --git a/packages/acp/src/index.ts b/packages/acp/src/index.ts index 283b8fc151..b418fad9f5 100644 --- a/packages/acp/src/index.ts +++ b/packages/acp/src/index.ts @@ -788,8 +788,15 @@ interface ResolvedResultPresentation { * both), the presenter remembers each `tool/call`'s `{ name, args }` keyed by * callId and looks it up on the matching result. The map is bridge-LOCAL (not a * change to the event schema or a core service): one presenter per live session - * (and a throwaway per `session/load` replay), entries removed as each result - * arrives, so it holds only the currently-in-flight calls. + * (and a throwaway per `session/load` replay), and each entry is removed when + * its result arrives. In the normal loop a `tool/call` is always followed by a + * `tool/result` (the registry turns even a thrown tool into an isError result), + * so the map holds only currently-in-flight calls. The one exception is a step + * torn down mid-tool (an abort between `tool/call` and `tool/result`), which can + * leave a single stale entry per such call; this is bounded by the session + * lifetime (the whole presenter is dropped on teardown) and never affects + * correctness — a later result for a different callId is unaffected, and the + * stale entry's only cost is one map slot until the session ends. */ export class ToolPresenter { private readonly pending = new Map() diff --git a/packages/acp/tests/harness.ts b/packages/acp/tests/harness.ts index 27335cfcff..4f6b5ac17a 100644 --- a/packages/acp/tests/harness.ts +++ b/packages/acp/tests/harness.ts @@ -18,6 +18,8 @@ import ToolRegistry from '@deepseek-ai/dsh-tools' import AgentRegistry from '@deepseek-ai/dsh-agent' import AgentLoop from '@deepseek-ai/dsh-agent-loop' import SessionPersistenceJsonl from '@deepseek-ai/dsh-session-persistence-jsonl' +import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local' +import * as ToolBash from '@deepseek-ai/dsh-tool-bash' import { ClientSideConnection, ndJsonStream, @@ -148,6 +150,14 @@ export async function makeBridgeHarness(options: { script?: (StreamChunk[] | 'hang')[] config?: Partial storageDir: string + /** + * Plug the REAL `dsh-bash-local` executor + `dsh-tool-bash` tools (instead of + * a test's own inline tool). Lets a test drive the actual `bash` tool — its + * real `presentCall`/`presentResult` — through the bridge, so tool-call UI + * tests verify the SHIPPING tool, not a stand-in (AGENTS.md "prefer the real + * implementation over a mock in tests"). + */ + withBash?: boolean } = { storageDir: '' }): Promise { const adapter = new MockAdapter(options.script ?? []) @@ -159,6 +169,10 @@ export async function makeBridgeHarness(options: { await ctx.plugin(AgentRegistry) await ctx.plugin(AgentLoop, { agents: [] }) await ctx.plugin(SessionPersistenceJsonl, { root: options.storageDir }) + if (options.withBash) { + await ctx.plugin(LocalBashExecutor, { timeoutMs: 10_000 }) + await ctx.plugin(ToolBash) + } ctx.llm.registerAdapter(['mock'], adapter) // Two identity byte pipes cross-wired into the two ndJsonStreams: bytes the diff --git a/packages/acp/tests/load.spec.ts b/packages/acp/tests/load.spec.ts index 3ffef1a87e..89bd8253f7 100644 --- a/packages/acp/tests/load.spec.ts +++ b/packages/acp/tests/load.spec.ts @@ -4,7 +4,6 @@ import { tmpdir } from 'node:os' import { join } from 'node:path' import { PROTOCOL_VERSION } from '@agentclientprotocol/sdk' import { SessionId } from '@deepseek-ai/dsh-session' -import { defineTool } from '@deepseek-ai/dsh-tools' import { makeBridgeHarness, textResponse, toolCallResponse, type BridgeHarness, type CapturedUpdate } from './harness.ts' /** Concatenate the text of all agent_message_chunk updates. */ @@ -58,64 +57,37 @@ describe('acp bridge — session/load replay', () => { }) it('replays a persisted tool call with the TOOL-OWNED presentation (title/rawInput/console output)', async () => { - // A turn with a tool call is persisted, then loaded by a fresh bridge. The - // replayed tool_call/tool_call_update must carry the tool's OWN presentation - // (presentCall/presentResult) — identical to how they streamed live — using - // a throwaway presenter that pairs call→result as the log replays in order. + // A turn with a REAL bash tool call is persisted, then loaded by a fresh + // bridge. The replayed tool_call/tool_call_update must carry the tool's OWN + // presentation — identical to how it streamed live — via a throwaway + // presenter that pairs call→result as the log replays in order. Uses the + // shipping tool (withBash), not a stand-in (AGENTS.md "prefer the real + // implementation over a mock in tests"). live = await makeBridgeHarness({ storageDir, - script: [toolCallResponse('c1', 'bash', { command: 'ls -la', description: 'List files' }), textResponse('done')], + withBash: true, + script: [toolCallResponse('c1', 'bash', { command: 'echo hello', description: 'Print a greeting' }), textResponse('done')], }) - live.ctx.tools.register(defineTool({ - name: 'bash', - description: 'run a command', - parameters: { - command: { type: 'string', required: true }, - description: { type: 'string', required: true }, - }, - async execute() { return [{ type: 'text', text: 'a.txt\nb.txt\n' }] }, - presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), - presentResult: (_args, result) => { - const block = result.content.length === 1 ? result.content[0] : undefined - if (block === undefined || block.type !== 'text') return undefined - return { content: [{ type: 'text', text: `\`\`\`console\n${block.text.trimEnd()}\n\`\`\`` }] } - }, - })) await live.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: {} }) const { sessionId } = await live.client.newSession({ cwd: process.cwd(), mcpServers: [] }) - await live.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'list' }] }) + await live.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'greet' }] }) await live.dispose() live = undefined - // A fresh bridge — which must ALSO have the tool registered, since the - // presentation is resolved from the live registry at replay time — loads it. - loader = await makeBridgeHarness({ storageDir, script: [] }) - loader.ctx.tools.register(defineTool({ - name: 'bash', - description: 'run a command', - parameters: { - command: { type: 'string', required: true }, - description: { type: 'string', required: true }, - }, - async execute() { return [] }, - presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), - presentResult: (_args, result) => { - const block = result.content.length === 1 ? result.content[0] : undefined - if (block === undefined || block.type !== 'text') return undefined - return { content: [{ type: 'text', text: `\`\`\`console\n${block.text.trimEnd()}\n\`\`\`` }] } - }, - })) + // A fresh bridge — also with the real bash tool, since the presentation is + // resolved from the live registry at replay time — loads the session. + loader = await makeBridgeHarness({ storageDir, withBash: true, script: [] }) await loader.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: {} }) await loader.client.loadSession({ sessionId, cwd: process.cwd(), mcpServers: [] }) const call = loader.updates.find(u => u.sessionUpdate === 'tool_call') - expect(call).toMatchObject({ toolCallId: 'c1', title: 'List files', kind: 'execute', rawInput: 'ls -la' }) + expect(call).toMatchObject({ toolCallId: 'c1', title: 'Print a greeting — echo hello', kind: 'execute', rawInput: 'echo hello' }) const update = loader.updates.find(u => u.sessionUpdate === 'tool_call_update') - expect(update).toMatchObject({ - toolCallId: 'c1', - status: 'completed', - content: [{ type: 'content', content: { type: 'text', text: '```console\na.txt\nb.txt\n```' } }], - }) + expect(update?.sessionUpdate).toBe('tool_call_update') + if (update?.sessionUpdate !== 'tool_call_update') throw new Error('expected a tool_call_update') + expect(update).toMatchObject({ toolCallId: 'c1', status: 'completed' }) + const content = update.content as { content: { text: string } }[] + expect(content[0]?.content.text).toBe('```console\nhello\n```') }) it('a load whose resume finishes after a client disconnect leaks no live session', async () => { diff --git a/packages/acp/tests/turns.spec.ts b/packages/acp/tests/turns.spec.ts index 101f6849fe..7b37233585 100644 --- a/packages/acp/tests/turns.spec.ts +++ b/packages/acp/tests/turns.spec.ts @@ -75,40 +75,41 @@ describe('acp bridge — turn outcomes', () => { expect(callIdx).toBeLessThan(updIdx) }) - it('a tool-owned presentation flows end-to-end: presentCall sets title/rawInput, presentResult reformats output', async () => { + it('the REAL bash tool drives the tool-call UI end-to-end: description—command title + console output', async () => { + // Use the SHIPPING tool (dsh-tool-bash + dsh-bash-local), not an inline + // stand-in, so this verifies the actual presentCall/presentResult the editor + // sees (AGENTS.md "prefer the real implementation over a mock in tests"). + // The mock MODEL still scripts the tool call (no real LLM needed), but the + // tool and executor are real: a real `echo` runs and its real output flows + // back through the bridge. harness = await makeBridgeHarness({ storageDir, - script: [toolCallResponse('c1', 'bash', { command: 'ls -la', description: 'List files' }), textResponse('done')], + withBash: true, + script: [ + toolCallResponse('c1', 'bash', { command: 'echo hello', description: 'Print a greeting' }), + textResponse('done'), + ], }) - // A tool that declares its OWN presentation (like the real tool-bash). The - // bridge must use it — NOT the generic title=name fallback — proving the - // tool-owns-its-rendering seam works through the real session-event path. - harness.ctx.tools.register(defineTool({ - name: 'bash', - description: 'run a command', - parameters: { - command: { type: 'string', required: true }, - description: { type: 'string', required: true }, - }, - async execute() { return [{ type: 'text', text: 'a.txt\nb.txt\n' }] }, - presentCall: args => ({ title: args.description, kind: 'execute', rawInput: args.command }), - presentResult: (_args, result) => { - const block = result.content.length === 1 ? result.content[0] : undefined - if (block === undefined || block.type !== 'text') return undefined - return { content: [{ type: 'text', text: `\`\`\`console\n${block.text.trimEnd()}\n\`\`\`` }] } - }, - })) const sessionId = await newSession(harness) - await harness.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'list' }] }) + await harness.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'greet' }] }) + // presentCall: execute kind, title is "description — command" (an execute + // card hides rawInput, so the command rides in the title), command in rawInput. const call = harness.updates.find(u => u.sessionUpdate === 'tool_call') - expect(call).toMatchObject({ toolCallId: 'c1', title: 'List files', kind: 'execute', rawInput: 'ls -la', status: 'in_progress' }) - const update = harness.updates.find(u => u.sessionUpdate === 'tool_call_update') - expect(update).toMatchObject({ + expect(call).toMatchObject({ toolCallId: 'c1', - status: 'completed', - content: [{ type: 'content', content: { type: 'text', text: '```console\na.txt\nb.txt\n```' } }], + title: 'Print a greeting — echo hello', + kind: 'execute', + rawInput: 'echo hello', + status: 'in_progress', }) + // presentResult: the REAL command output, wrapped in a fenced console block. + const update = harness.updates.find(u => u.sessionUpdate === 'tool_call_update') + expect(update?.sessionUpdate).toBe('tool_call_update') + if (update?.sessionUpdate !== 'tool_call_update') throw new Error('expected a tool_call_update') + expect(update).toMatchObject({ toolCallId: 'c1', status: 'completed' }) + const content = update.content as { content: { type: string; text: string } }[] + expect(content[0]?.content.text).toBe('```console\nhello\n```') }) it('a throwing tool presenter does not break the turn: the bridge falls back generically', async () => { diff --git a/packages/tool-bash/README.md b/packages/tool-bash/README.md index 473f5a4099..194c94c1ca 100644 --- a/packages/tool-bash/README.md +++ b/packages/tool-bash/README.md @@ -34,7 +34,7 @@ The owning agent is recorded per task id at spawn and kept for the lifetime of t ## UI presentation -These tools own how their calls render in a UI (an editor's tool-call card) via the `dsh-tools` `presentCall`/`presentResult` seam — a UI never special-cases tool names. For `bash`: the model-written `description` is the always-visible **title** (e.g. "List files in the current directory"), the exact `command` is the **rawInput** (the verbatim command stays visible in a detail view without crowding the title), `kind` is `execute` (terminal/run treatment), and the completed output is wrapped in a fenced ` ```console ` block — a UI-only affordance, so the model-facing result text stays unfenced. `bash_output`/`bash_kill` present a task-scoped title ("Read output from background task bash-3" / "Kill background task bash-3") with the task id as rawInput. These methods are pure/display-only (they also run on `session/load` replay), and a malformed/older logged arg shape falls back to a generic presentation rather than throwing. See `packages/tools` ("Tool-owned UI presentation") and `packages/acp` ("Tool-call presentation"). +These tools own how their calls render in a UI (an editor's tool-call card) via the `dsh-tools` `presentCall`/`presentResult` seam — a UI never special-cases tool names. For `bash`: the **title** is the model-written `description` followed by the exact `command` ("List files in src — ls -la src"), `kind` is `execute` (terminal/run treatment), and the `command` is ALSO the **rawInput**. Why both in the title: an execute-kind card hides `rawInput` (Zed renders it only for non-terminal tools), so the command must ride in the always-visible title to be seen — the reference ACP adapters (claude-agent-acp, codex-acp) likewise put the command in an execute tool's title. The completed output is wrapped in a fenced ` ```console ` block — a UI-only affordance, so the model-facing result text stays unfenced. `bash_output`/`bash_kill` present a task-scoped title ("Read output from background task bash-3" / "Kill background task bash-3") with the task id as rawInput. These methods are pure/display-only (they also run on `session/load` replay), and a malformed/older logged arg shape falls back to a generic presentation rather than throwing. See `packages/tools` ("Tool-owned UI presentation") and `packages/acp` ("Tool-call presentation"). ## Background completion notices diff --git a/packages/tool-bash/src/index.ts b/packages/tool-bash/src/index.ts index 9b4aa0819b..660c4ecfea 100644 --- a/packages/tool-bash/src/index.ts +++ b/packages/tool-bash/src/index.ts @@ -133,15 +133,18 @@ export function renderResult(result: BashRunResult): string { // --------------------------------------------------------------------------- /** - * Pending-state presentation for a `bash` call: the model-written `description` - * is the always-visible title (the schema requires it precisely so a UI has a - * readable summary — "List files in the current directory"), `kind: 'execute'` - * (a terminal/run treatment), and the exact `command` is the `rawInput` so the - * verbatim command stays visible in a UI's detail view without crowding the - * title. Mirrors how Zed / the reference ACP adapters render execute tools. + * Pending-state presentation for a `bash` call. The title is the model-written + * `description` followed by the exact `command` ("List files — ls -la src"): + * `kind: 'execute'` gets a terminal/run treatment in a UI, but an execute-kind + * card HIDES `rawInput` (Zed: `should_show_raw_input = !is_terminal_tool`), so + * the command MUST ride in the always-visible title to be seen — the reference + * ACP adapters (claude-agent-acp, codex-acp) likewise put the command in the + * title for execute tools. The description leads (a readable summary the schema + * requires); the command follows so the verbatim text is still there. `rawInput` + * still carries the bare command for non-execute UIs that DO render it. */ function presentBashCall(args: { command: string; description: string }): ToolCallPresentation { - return { title: args.description, kind: 'execute', rawInput: args.command } + return { title: `${args.description} — ${args.command}`, kind: 'execute', rawInput: args.command } } /** diff --git a/packages/tool-bash/tests/tools.spec.ts b/packages/tool-bash/tests/tools.spec.ts index 8b68de213b..c1b1e18808 100644 --- a/packages/tool-bash/tests/tools.spec.ts +++ b/packages/tool-bash/tests/tools.spec.ts @@ -564,10 +564,10 @@ describe('status lines', () => { }) describe('tool-owned UI presentation (presentCall / presentResult)', () => { - it('bash presentCall: the model description is the title, the command is the rawInput, kind execute', async () => { + it('bash presentCall: title is "description — command" (execute cards hide rawInput), command also in rawInput', async () => { const ctx = await setup() - const present = ctx.tools.get('bash')!.presentCall!({ command: 'ls -la src', description: 'List files in src' }) - expect(present).toEqual({ title: 'List files in src', kind: 'execute', rawInput: 'ls -la src' }) + const present = ctx.tools.get('bash')?.presentCall?.({ command: 'ls -la src', description: 'List files in src' }) + expect(present).toEqual({ title: 'List files in src — ls -la src', kind: 'execute', rawInput: 'ls -la src' }) }) it('bash presentResult: wraps the model-facing text in a fenced console block', async () => { diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 9c80db6c8f..9c534153ab 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -84,6 +84,9 @@ importers: '@deepseek-ai/dsh-agent-loop': specifier: workspace:^ version: link:../agent-loop + '@deepseek-ai/dsh-bash-local': + specifier: workspace:^ + version: link:../bash-local '@deepseek-ai/dsh-llm': specifier: workspace:^ version: link:../llm @@ -99,6 +102,9 @@ importers: '@deepseek-ai/dsh-system-prompt': specifier: workspace:^ version: link:../system-prompt + '@deepseek-ai/dsh-tool-bash': + specifier: workspace:^ + version: link:../tool-bash '@deepseek-ai/dsh-tools': specifier: workspace:^ version: link:../tools