Merge remote-tracking branch 'origin/master' into feat/code-runtime-multilang-seam
# Conflicts: # docs/core-data-structures/code-runtime.i18n.yaml # packages/code-runtime/code-runtime/README.i18n.yaml
This commit is contained in:
@@ -326,6 +326,8 @@
|
||||
|
||||
- id: tool-todo
|
||||
name: '@deepseek-ai/dsh-tool-todo'
|
||||
config:
|
||||
allowParallelInProgress: true
|
||||
|
||||
# Persisted same-session goals reach the model and the slash menu here; the
|
||||
# domain, driver, and `/goal` command are above.
|
||||
|
||||
@@ -474,11 +474,14 @@ function buildAlphaLog(): SessionEvent[] {
|
||||
push({ type: 'step/end', data: { turn, step: 0 } })
|
||||
push({ type: 'turn/end', data: { turn, reason: { kind: 'completed' } } })
|
||||
}
|
||||
// Turn 67: todo_write sample — the TodoRow toolview in the flow plus the
|
||||
// todo/write snapshot event feeding the TodoPanel plan strip.
|
||||
// Turn 71: todo_write sample — the TodoRow toolview in the flow plus the
|
||||
// todo/write snapshot event feeding the TodoPanel plan strip. Two items are
|
||||
// in_progress: this fixture chooses the parallel policy, so both surfaces
|
||||
// must render a parallel plan rather than the first active item alone.
|
||||
const fixtureTodos = [
|
||||
{ content: '梳理需求', status: 'completed' },
|
||||
{ content: '实现 fixture 样本', status: 'in_progress' },
|
||||
{ content: '跑后台构建', status: 'in_progress' },
|
||||
{ content: '浏览器验收', status: 'pending' },
|
||||
]
|
||||
// Turn 65: the terminal sample turn 60's two clean prompt rows cannot cover —
|
||||
@@ -531,7 +534,7 @@ function buildAlphaLog(): SessionEvent[] {
|
||||
toolTurn(70, 'web_fetch', '{"url":"https://www.deepseek.com/blog/harness-architecture"}', '# Harness architecture\n\nEverything is a plugin.')
|
||||
|
||||
const todoArgs = JSON.stringify({ todos: fixtureTodos })
|
||||
toolTurn(71, 'todo_write', todoArgs, 'Updated todo list: 1 pending, 1 in progress, 1 completed.')
|
||||
toolTurn(71, 'todo_write', todoArgs, 'Updated todo list: 1 pending, 2 in progress, 1 completed.')
|
||||
// The real tool appends the snapshot mid-execution — between tool/call and
|
||||
// tool/result — so the fixture reproduces that exact ordering (the last
|
||||
// toolTurn events run ... tool/call, tool/result, step/end, turn/end).
|
||||
|
||||
@@ -242,6 +242,10 @@ describe('createFixtureApi', () => {
|
||||
const times = events.slice(todoAt - 1, todoAt + 2).map(e => e.time)
|
||||
expect(times[0]).toBeLessThanOrEqual(times[1] ?? 0)
|
||||
expect(times[1]).toBeLessThanOrEqual(times[2] ?? 0)
|
||||
// The sample is a parallel plan: this fixture chooses the parallel policy,
|
||||
// so the surfaces fed from here face more than one active item.
|
||||
const snapshot = events[todoAt] as { data: { todos: { status: string }[] } }
|
||||
expect(snapshot.data.todos.filter(t => t.status === 'in_progress')).toHaveLength(2)
|
||||
})
|
||||
|
||||
it('create adds a session and pushes host/session-added to open host streams', async () => {
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/client/ui-conversation/README.md
|
||||
README.md: bbd115eac0eb914914dc11e504639633c801abdd
|
||||
README.zh.md: 843b49e311fbf1a9157413c42a0ef3e9828284bc
|
||||
README.md: a75f25d8669cd688795842a655106e0e27bb7173
|
||||
README.zh.md: f0d744c31020730857d210d75749b907dbffca08
|
||||
|
||||
@@ -34,11 +34,11 @@ A `grep`/`glob` call declaring the `search` render intent renders its result inl
|
||||
|
||||
Tool rows use the keyed, session-scoped `'conversation.chat.toolview'` slot; its render site dispatches via `entryKey: toolName` with `GenericToolCard` as the call-site fallback. The owner payload is the uniform `ToolRowOwnerProps` (`callId`/`toolName`/`block`/`openFile`), and `ToolRowProps` composes it with the session standard kit. A registrant is a plain plugin with only the slot service edge: `ctx.slots.inject('conversation.chat.toolview', () => ctx.slots.register({ name: 'conversation.chat.toolview', key: '<tool>', inject? }, Row))`. The declaration is the activation and reload dependency; `ConversationService` is required only by registrations that call its actions. Trajectory and waterfall toolview slots share this shape and use their own render sites; RendersCheck rejects a declaration nobody renders.
|
||||
|
||||
The todo surfaces are two registrations over that shape, both using slot declaration injection without a `ConversationService` edge. `TodoRow` takes the `'conversation.chat.toolview'` key `todo_write` and summarizes what the call attempted (`<done>/<total> 已完成 · <active item>` parsed from its args, falling back to the generic summary on malformed or wrongly-shaped model JSON, and keeping the generic dot for non-ok execution states so a cancelled call never reads as a completed update). `TodoDock` takes the `'conversation.input.dock'` list slot at `order: 0` — before Goal and Queue — and is the plan strip: it reads the host-computed `todos` projection via `useProjection` (standing plan: latest `todo/write` with no later `turn/start`) and renders `TodoPanel`, which takes the plain list, hides itself while the list is empty, and starts collapsed as a header of title plus `"<done>/<total> tasks · <n> in progress"` (status glyphs are the figma check / progress / dashed-pending set). The dock adapter owns the selection so the panel stays a pure function of its props; the standing list lives here rather than in the row so the row stays one line. Anything the input-zone composer chain hides (a `conversation.composer` takeover such as ui-question's) hides the whole dock, this strip included.
|
||||
The todo surfaces are two registrations over that shape, both using slot declaration injection without a `ConversationService` edge. `TodoRow` takes the `'conversation.chat.toolview'` key `todo_write` and summarizes what the call attempted (`<done>/<total> completed · <active item>` plus a `+<n>` count of the other active ones, parsed from its args through `toolviews/plan-summary.ts` `planSummary`, falling back to the generic summary on malformed or wrongly-shaped model JSON, and keeping the generic dot for non-ok execution states so a cancelled call never reads as a completed update). When the deployment permits parallel work, several items may be `in_progress` at once, so `planSummary` names the first and counts the rest, and deliberately returns the two unjoined: the row ellipsizes its summary text, so a count concatenated onto the end of the task name would be the first thing a narrow row clips. The row hands the count to `ToolRow`'s `summarySuffix`, the shared row's non-shrinking slot beside that ellipsized text (an error row drops it, since its collapsed summary is the failure line). `TodoDock` takes the `'conversation.input.dock'` list slot at `order: 0` — before Goal and Queue — and is the plan strip: it reads the host-computed `todos` projection via `useProjection` (standing plan: latest `todo/write` with no later `turn/start`) and renders `TodoPanel`, which takes the plain list, hides itself while the list is empty, and starts collapsed as a header of title plus its own `·`-joined per-status counts (localized, `1 completed · 2 in progress · 1 pending`, zero-count segments omitted; status glyphs are the figma check / progress / dashed-pending set), so it reports the parallel count without needing a name to truncate. The dock adapter owns the selection so the panel stays a pure function of its props; the standing list lives here rather than in the row so the row stays one line. Anything the input-zone composer chain hides (a `conversation.composer` takeover such as ui-question's) hides the whole dock, this strip included.
|
||||
|
||||
`QueueDock` is the terminal input-dock entry at `order: 20`. It hides while empty, renders one pending row directly, and defaults two or more rows to a collapsed `"<n> 条排队消息"` header whose button expands or collapses the complete list. The header exposes `aria-expanded` and `aria-controls`; the expanded list scrolls within a 180px height bound. An active edit or mutation keeps its rows visible, and emptying the queue restores the collapsed default for the next queue. Each visible ordinary-session row remains a single-line preview with its exact-occurrence edit, delete, and strict-steer actions; addressed subagents retain the rows as a read-only projection because their continuation transport does not expose queue mutation. If strict steer loses to a closed window, the original occurrence remains queued for normal delivery; if the driver already claimed it, normal delivery is already underway. Neither converged race displays a failure, while transport and unknown failures do.
|
||||
|
||||
The Host's placement-aware `session/queue` snapshot also carries pending steering. QueueDock filters it out, while ChatView projects it as a user-style bubble with Copy at the conversation tail; non-user next-step items (injected context) carry the `context` placement instead and render nowhere until claimed. Fork stays absent because the message has not entered a durable turn. The Host delays steering retirement until the durable `user/message` carrying the steering has entered the mux stream. On that accepted live event, the client runtime retires the first matching current steering occurrence before publishing the snapshot; historical events cannot hide later occurrences that reuse the same `MessageId`. The bubble therefore hands off without a gap or duplicate, immediately restores Copy and the branch control from the durable node, enables branch only when that node is the completed turn's transcript tail, and survives reconnect from the same authority.
|
||||
The Host's placement-aware `session/queue` snapshot also carries pending steering. QueueDock filters it out, while ChatView projects it as a user-style bubble with Copy at the conversation tail; non-user next-step items (injected context) carry the `context` placement instead and render nowhere until claimed. Fork is absent here as on every user-style bubble. The Host delays steering retirement until the durable `user/message` carrying the steering has entered the mux stream. On that accepted live event, the client runtime retires the first matching current steering occurrence before publishing the snapshot; historical events cannot hide later occurrences that reuse the same `MessageId`. The bubble therefore hands off without a gap or duplicate, immediately restores Copy and the clock from the durable node — a steering bubble, like a user bubble, carries no branch action ([decision](../../../.agents/notes/implemented/simplification/2026-08-06-user-bubbles-drop-the-branch-action.md)) — and survives reconnect from the same authority.
|
||||
|
||||
Keyboard message submission resolves delivery from the addressed session's running state and steering capability. While idle, Enter and Cmd/Ctrl+Enter both perform an ordinary Queue send. While a primary session is running, the browser-persisted General Settings preference assigns plain Enter to `Queue` (the default) or `Steer`, and Cmd/Ctrl+Enter performs the other behavior; Shift+Enter remains a newline. Addressed subagents keep both gestures on their Queue-only continuation transport even while running. The preference affects only the steer-capable busy-state gesture pair, and the send button and non-keyboard submit actions remain Queue. Composer Steer uses the existing best-effort `session.prompt(mode: 'steer')` contract: if the current next-step window closes before acceptance, AgentLoop admits the message as the next waking Queue turn without surfacing a failure or losing the draft transaction.
|
||||
|
||||
@@ -63,8 +63,8 @@ None; this package neither assembles nor sends a provider request.
|
||||
- **Compaction markers show no scale** — the row does not yet report how many messages or which range the checkpoint replaced.
|
||||
- **Stats-line durations and speeds cover the in-window flow only** — LLM and tool wall times plus the TTFT and throughput averages fold the snapshot's assistant `timing` and tool call/result pairs, so nodes outside the loaded event window (older history) are not counted.
|
||||
- **The details panel has no entry point** — `ChatViewInjected.openDetails` is implemented but uncalled, so the raw selected-call display is unreachable in the assembled application. There is no Input/Output/Metadata switch, Prev/Next stepping, or trajectory deep link.
|
||||
- **Assistant per-message paging is a reserved slot** — drawn in the design, not implemented. The finalized content IconActions row (copy / clock / branch) ships under the last content-text assistant of each turn only; mid-turn narration and Think-only nodes stay chrome-free. Branch stays disabled unless that message is also the last transcript node of a completed turn; when enabled, it forks through that turn, increments the inherited title on the client, and opens the child. A fork or rename failure leaves the source selected ([decision](../../../.agents/notes/implemented/bug-fix/2026-08-02-message-fork-actions-require-completed-turn-tail.md)).
|
||||
- **Sent user messages cannot be edited** — user bubbles retain clock, copy, and branch; branch stays disabled unless a completed turn's transcript ends at that user message. Editing returns with the capability behind it: a client mutation over a settled user message, plus the host behavior for the turn that already consumed it ([decision](../../../.agents/notes/implemented/simplification/2026-07-31-drop-user-message-edit-stub.md)).
|
||||
- **Assistant per-message paging is a reserved slot** — drawn in the design, not implemented. The finalized content IconActions row (copy / clock / branch) ships under the last content-text assistant of each turn that has ended; mid-turn narration, Think-only nodes, and every node of a turn still producing steps stay chrome-free. Branch stays disabled unless that message is also the last transcript node of a completed turn; when enabled, it forks through that turn, increments the inherited title on the client, and opens the child. A fork or rename failure leaves the source selected ([decision](../../../.agents/notes/implemented/bug-fix/2026-08-02-message-fork-actions-require-completed-turn-tail.md)).
|
||||
- **Sent user messages cannot be edited** — user bubbles retain clock and copy; branch lives only under assistant answers ([decision](../../../.agents/notes/implemented/simplification/2026-08-06-user-bubbles-drop-the-branch-action.md)). Editing returns with the capability behind it: a client mutation over a settled user message, plus the host behavior for the turn that already consumed it ([decision](../../../.agents/notes/implemented/simplification/2026-07-31-drop-user-message-edit-stub.md)).
|
||||
- **The sparkle icon for the others tool row is a hand-drawn approximation** — the design glyph's vector geometry is not exportable locally; promotion into ui-primitives waits on an exact export.
|
||||
- **The approval panel has no durable grant control** — it supports allow-once and reject only.
|
||||
- **TodoPanel truncates long item text to one ellipsized line** — the figma strip has no wrap or expand affordance; full text is not readable inline.
|
||||
|
||||
@@ -34,11 +34,11 @@ Think 行默认保持折叠,并在不展开思维链的情况下暴露实时
|
||||
|
||||
审批经由本包声明的链接管编辑器:`ApprovalPanel` 注册为按选择器路由的 `'conversation.composer'` 配置项(ui-question 模式),在审批等待未决期间取代 InputBar 占据编辑器(琥珀色条、理由标题、来自运行中调用参数的配对命令行、一次性的拒绝/允许)。`contract/slots.ts` 中的 `PendingApproval` 领域面在运行时 `PendingWait` 载体之上拥有 wire 编码——带审计关联的 `ApprovalResponsePayload` 值;广播的 `approval/resolved` 帧使等待落定并恢复编辑器。运行时 manager 会将所有审批或问题等待通过 `SessionSummary.pendingInteraction` 投影出来,未实例化的 Session 也不例外;`ui-workspace` 负责其侧边栏呈现。未决等待完全离开消息流:问题(ui-question)与审批(ApprovalPanel)都经编辑器接管作答,不再保留只读占位卡。编辑器底行的 Access 席位挂载 `PermissionSelect`,由 host 计算的 `permissions` 投影经标准工具包 `useProjection` 供数(key 缺席即隐藏 chip);chip 打开 Menu 原语下拉,其中 kebab-case 预设名渲染为 Title Case 标签;普通安全预设会立即经输入栏注入的 `command` 回调提交 `/permission <preset>`,而 `danger-full-access` 在界面中显示为 `Full access`,选择后先打开页面内的 Modal 风险确认。用户勾选确认项前启用按钮始终不可用;取消、Escape、关闭按钮与点击遮罩都不会提交命令。
|
||||
|
||||
todo 两个面就是在该形状上的两个注册项,都使用 slot 声明注入,不依赖 `ConversationService`。`TodoRow` 占用 `'conversation.chat.toolview'` 的 `todo_write` key,摘要该次调用「试图写入」的内容(从其 args 解析出 `<已完成>/<总数> 已完成 · <进行中条目>`;模型 JSON 残缺或形状不对时回落到通用摘要;非 ok 执行状态保留通用状态点,使被取消的调用绝不读成一次已完成的更新)。`TodoDock` 以 `order: 0` 占用 `'conversation.input.dock'` 列表 slot(位于 Goal 与 Queue 之前),是计划条:它经 `useProjection` 读取 host 计算的 `todos` 投影(站立计划:其后没有更晚 `turn/start` 的最近一次 `todo/write`)并渲染 `TodoPanel`,后者接收纯列表,在列表为空时自我隐藏;列表非空时面板初始折叠,表头显示标题加 `"<已完成>/<总数> tasks · <n> in progress"`(状态图标为 figma 的勾选/进行中/虚线未开始一组)。选取由 dock 适配器负责,因此面板保持为其 props 的纯函数;站立列表放在此处而非行内,行才能保持单行。输入区 composer 链隐藏的一切(例如 ui-question 对 `conversation.composer` 的接管)也会隐藏整个 dock,包括这条计划条。
|
||||
todo 两个面就是在该形状上的两个注册项,都使用 slot 声明注入,不依赖 `ConversationService`。`TodoRow` 占用 `'conversation.chat.toolview'` 的 `todo_write` key,摘要该次调用「试图写入」的内容(从其 args 经 `toolviews/plan-summary.ts` 的 `planSummary` 解析出 `<已完成>/<总数> 已完成 · <进行中条目>`,以及「其余活跃项的数量」`+<n>`;模型 JSON 残缺或形状不对时回落到通用摘要;非 ok 执行状态保留通用状态点,使被取消的调用绝不读成一次已完成的更新)。部署允许并行工作时,可以有多个条目同时处于 `in_progress`,因此 `planSummary` 给出第一个活跃条目并计数其余,且刻意不把两者拼成一个字符串:行会对摘要文本做省略号截断,把数量接在任务名末尾时,窄行最先裁掉的正是这个数量。该行把数量交给 `ToolRow` 的 `summarySuffix`——共享行在被截断文本旁的不收缩位(出错的行会丢弃它,因为其折叠摘要是失败首行)。`TodoDock` 以 `order: 0` 占用 `'conversation.input.dock'` 列表 slot(位于 Goal 与 Queue 之前),是计划条:它经 `useProjection` 读取 host 计算的 `todos` 投影(站立计划:其后没有更晚 `turn/start` 的最近一次 `todo/write`)并渲染 `TodoPanel`,后者接收纯列表,在列表为空时自我隐藏;列表非空时面板初始折叠,表头显示标题加它自行计算的、以 `·` 连接的各状态计数(本地化,形如 `1 已完成 · 2 进行中 · 1 待处理`,计数为零的段落省略;状态图标为 figma 的勾选/进行中/虚线未开始一组),因此它无需一个可被截断的任务名即可报告并行数量。选取由 dock 适配器负责,因此面板保持为其 props 的纯函数;站立列表放在此处而非行内,行才能保持单行。输入区 composer 链隐藏的一切(例如 ui-question 对 `conversation.composer` 的接管)也会隐藏整个 dock,包括这条计划条。
|
||||
|
||||
`QueueDock` 是 `order: 20` 的末端 input-dock 条目。队列为空时隐藏;只有一个待处理项时直接渲染该行;存在两个或更多待处理项时,默认收起为 `"<n> 条排队消息"` 表头,其按钮可展开或收起完整列表。表头暴露 `aria-expanded` 和 `aria-controls`;展开后的列表以 180px 为高度上限,并可滚动。存在进行中的编辑或变更时,列表行会保持可见;队列清空后,下一次出现队列时会恢复默认收起状态。普通会话中的每条可见行仍是单行预览,并提供针对精确单次入队项的编辑、删除和严格 steering 操作;已寻址 subagent 则保留只读行,因为其继续执行传输不提供 Queue 变更。如果严格 steering 输给已关闭的窗口,原单次入队项会留在 Queue 中正常投递;如果驱动器已经认领该项,正常投递就已开始。这两种已收敛的竞态都不显示失败,传输和未知错误仍会显示。
|
||||
|
||||
Host 带 placement 的 `session/queue` 快照也会携带待处理 steering。QueueDock 会将其过滤掉,ChatView 则把它投影为会话流末尾带复制操作的用户样式气泡;非用户来源的 next-step 项(注入上下文)改以 `context` placement 广播,领取前不在任何界面渲染。消息尚未进入持久轮次,因此不显示 fork。Host 会等携带该 steering 的持久 `user/message` 进入 mux 流之后再退役 steering。客户端运行时接纳该实时事件时,会在发布快照前退役第一个匹配的当前 steering 单次入队项;历史事件无法隐藏后来复用同一 `MessageId` 的单次入队项。气泡交接时因而不会产生空档或重复,会立即从持久节点恢复复制操作与分支控件,仅当该节点是已完成轮次的 transcript 尾部时才启用分支,并能在重连后从同一权威恢复。
|
||||
Host 带 placement 的 `session/queue` 快照也会携带待处理 steering。QueueDock 会将其过滤掉,ChatView 则把它投影为会话流末尾带复制操作的用户样式气泡;非用户来源的 next-step 项(注入上下文)改以 `context` placement 广播,领取前不在任何界面渲染。与所有用户样式气泡一样,这里不显示 fork。Host 会等携带该 steering 的持久 `user/message` 进入 mux 流之后再退役 steering。客户端运行时接纳该实时事件时,会在发布快照前退役第一个匹配的当前 steering 单次入队项;历史事件无法隐藏后来复用同一 `MessageId` 的单次入队项。气泡交接时因而不会产生空档或重复,会立即从持久节点恢复复制操作与时钟——steering 气泡与 user 气泡一样不带分支操作([决策](../../../.agents/notes/implemented/simplification/2026-08-06-user-bubbles-drop-the-branch-action.md))——并能在重连后从同一权威恢复。
|
||||
|
||||
键盘消息提交会根据所寻址会话的运行状态和 steering 能力解析投递方式。空闲时,Enter 和 Cmd/Ctrl+Enter 都执行普通 Queue 发送。主会话运行期间,浏览器持久化的 General Settings 偏好会把普通 Enter 分配为 `Queue`(默认值)或 `Steer`,Cmd/Ctrl+Enter 则执行另一种行为;Shift+Enter 仍然换行。已寻址 subagent 即使正在运行,也会让这两个手势都使用其仅支持 Queue 的继续执行传输。该偏好只影响支持 steering 的繁忙态手势对,发送按钮与非键盘提交操作仍使用 Queue。Composer Steer 复用现有尽力而为的 `session.prompt(mode: 'steer')` 契约:如果当前 next-step 窗口在接纳前关闭,AgentLoop 会把消息接纳为下一条唤醒 Queue 轮次,不显示失败,也不会丢失草稿事务。
|
||||
|
||||
@@ -63,8 +63,8 @@ Host 带 placement 的 `session/queue` 快照也会携带待处理 steering。Qu
|
||||
- **压缩标记不显示规模**:该行尚不报告检查点替换了多少条消息或哪段范围。
|
||||
- **统计行的耗时与速率只覆盖窗口内消息流**:LLM 与工具墙钟时间以及 TTFT 与吞吐平均值由快照的 assistant `timing` 与工具 call/result 配对折算,落在已加载事件窗口之外的节点(更早的历史)不计入。
|
||||
- **详情面板没有入口**:`ChatViewInjected.openDetails` 虽已实现却无人调用,因此以原始形式显示已选择调用的那部分在组装后的应用中不可达。没有 Input/Output/Metadata 切换、Prev/Next 步进,也没有 trajectory 深链接。
|
||||
- **assistant 逐消息分页是预留 slot**:设计中已有图稿,尚未实现。已定稿的内容 IconActions 行(复制/时钟/分支)只挂在每个轮次中最后一条带 text 内容的 assistant 下;轮次中间的叙述与纯 Think 节点不带 chrome。除非该消息同时也是已完成轮次的最后一个 transcript 节点,否则分支保持禁用;启用后,它会 fork 到该轮次末尾,在 client 端递增继承标题并打开子会话。fork 或改名失败时源会话保持选中([决策](../../../.agents/notes/implemented/bug-fix/2026-08-02-message-fork-actions-require-completed-turn-tail.md))。
|
||||
- **已发送的 user 消息无法编辑**:user 气泡保留时钟、复制和分支;除非已完成轮次的 transcript 结束于该 user 消息,否则分支保持禁用。编辑功能要与其背后的能力一起回归:既需要针对已定稿 user 消息的 client 变更,也需要 host 侧对已经消费过它的轮次给出行为([决策](../../../.agents/notes/implemented/simplification/2026-07-31-drop-user-message-edit-stub.md))。
|
||||
- **assistant 逐消息分页是预留 slot**:设计中已有图稿,尚未实现。已定稿的内容 IconActions 行(复制/时钟/分支)只挂在每个已结束轮次中最后一条带 text 内容的 assistant 下;轮次中间的叙述、纯 Think 节点,以及仍在产出步骤的轮次里的所有节点都不带 chrome。除非该消息同时也是已完成轮次的最后一个 transcript 节点,否则分支保持禁用;启用后,它会 fork 到该轮次末尾,在 client 端递增继承标题并打开子会话。fork 或改名失败时源会话保持选中([决策](../../../.agents/notes/implemented/bug-fix/2026-08-02-message-fork-actions-require-completed-turn-tail.md))。
|
||||
- **已发送的 user 消息无法编辑**:user 气泡保留时钟和复制;分支只存在于 assistant 回答之下([决策](../../../.agents/notes/implemented/simplification/2026-08-06-user-bubbles-drop-the-branch-action.md))。编辑功能要与其背后的能力一起回归:既需要针对已定稿 user 消息的 client 变更,也需要 host 侧对已经消费过它的轮次给出行为([决策](../../../.agents/notes/implemented/simplification/2026-07-31-drop-user-message-edit-stub.md))。
|
||||
- **others 工具行的闪光图标是手绘近似版本**:无法在本地导出设计字形的矢量几何;等到存在精确导出后再将其提升到 ui-primitives。
|
||||
- **审批面板的「始终允许此类」暂缓**:持久授权需要授权存储设计;今天只能回答允许一次/拒绝。
|
||||
- **TodoPanel 将过长条目截成单行省略号**:figma 条没有换行或展开入口,完整文本无法在行内读完。
|
||||
|
||||
@@ -4,10 +4,10 @@
|
||||
// view groups them into tool rows through its keyed toolview slot (figma
|
||||
// step-summary flow). Shared by finalized nodes and the streaming partial;
|
||||
// the turn-level loading dots live in the chat view's tail, not here.
|
||||
// Finalized content (text) nodes append IconActions once streaming ends
|
||||
// (`time` is omitted for mid-turn narration); their branch action is enabled
|
||||
// only when the node is also the completed turn's transcript tail. Think /
|
||||
// tool-head-only nodes stay chrome-free.
|
||||
// Finalized content (text) nodes append IconActions once their turn ends
|
||||
// (`time` is omitted for mid-turn narration and while the turn still runs);
|
||||
// their branch action is enabled only when the node is also the completed
|
||||
// turn's transcript tail. Think / tool-head-only nodes stay chrome-free.
|
||||
|
||||
import { memo, useMemo } from 'react'
|
||||
import type { AssistantBlock } from '@deepseek-ai/dsh-client-runtime/client'
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
IconThinkOutline14, JsonBlock, MarkdownText,
|
||||
} from '@deepseek-ai/dsh-client-ui-primitives'
|
||||
import type { ChatViewSlotProps } from '../contract/slots.ts'
|
||||
import { hasContentText } from './chat-flow.ts'
|
||||
import { MessageIconActions } from './MessageIconActions.tsx'
|
||||
import { ToolRow } from './ToolRow.tsx'
|
||||
import css from './AssistantMarkdown.module.css'
|
||||
@@ -25,7 +26,8 @@ export interface AssistantMarkdownProps {
|
||||
/** Frozen partial of an aborted turn: rendered with a stopped marker. */
|
||||
interrupted?: boolean | undefined
|
||||
/** Unix epoch ms for the IconActions clock; omitted while streaming or when
|
||||
* the parent withholds chrome (mid-turn content assistants). */
|
||||
* the parent withholds chrome (mid-turn content assistants and every node
|
||||
* of a turn that has not ended). */
|
||||
time?: number | undefined
|
||||
/** Turn wall time in ms for the IconActions run-time label; omitted when the
|
||||
* turn's triggering input is outside the loaded window. */
|
||||
@@ -65,11 +67,6 @@ function copyText(blocks: readonly AssistantBlock[]): string {
|
||||
return parts.join('')
|
||||
}
|
||||
|
||||
/** True when the node has model-visible text content worth chrome under. */
|
||||
function hasContentText(blocks: readonly AssistantBlock[]): boolean {
|
||||
return blocks.some(block => block.kind === 'text' && block.text.trim() !== '')
|
||||
}
|
||||
|
||||
/** Reasoning block as the Think variant summary row (figma 39:28304). */
|
||||
function ThinkRow({ text, running, t }: { text: string; running: boolean; t: AssistantMarkdownProps['t'] }) {
|
||||
return (
|
||||
|
||||
@@ -30,7 +30,7 @@ import type {
|
||||
import type { SnapshotSelectorHook } from '@deepseek-ai/dsh-client-ui-slots'
|
||||
import { IconChevronDownOutline14 } from '@deepseek-ai/dsh-client-ui-primitives'
|
||||
import type { ChatViewSlotProps } from '../contract/slots.ts'
|
||||
import { assistantActionsSeqs, deriveChatFlow, messageBranchSeqs, runningTurnStartTime, type ChatFlowItem } from './chat-flow.ts'
|
||||
import { assistantActionsSeqs, assistantBranchSeqs, deriveChatFlow, runningTurnStartTime, type ChatFlowItem } from './chat-flow.ts'
|
||||
import { AssistantMarkdown } from './AssistantMarkdown.tsx'
|
||||
import { GenericCommandCard } from './GenericCommandCard.tsx'
|
||||
import { GenericToolCard } from './GenericToolCard.tsx'
|
||||
@@ -358,10 +358,11 @@ export function ChatView({
|
||||
[inbox],
|
||||
)
|
||||
const activeRetry = useMemo(() => activeRetrySeq(nodes, running), [nodes, running])
|
||||
// Only the last content assistant of each turn owns IconActions; mid-turn
|
||||
// text (before tools) omits `time` so AssistantMarkdown stays chrome-free.
|
||||
const actionSeqs = useMemo(() => assistantActionsSeqs(nodes), [nodes])
|
||||
const branchSeqs = useMemo(() => messageBranchSeqs(nodes, turnEnds), [nodes, turnEnds])
|
||||
// Only the last content assistant of each completed turn owns IconActions;
|
||||
// mid-turn text and every node of a running turn omit `time`, so
|
||||
// AssistantMarkdown stays chrome-free until the answer settles.
|
||||
const actionSeqs = useMemo(() => assistantActionsSeqs(nodes, turnEnds), [nodes, turnEnds])
|
||||
const branchSeqs = useMemo(() => assistantBranchSeqs(nodes, turnEnds), [nodes, turnEnds])
|
||||
const runningTurnStart = useMemo(() => runningTurnStartTime(turnTimings), [turnTimings])
|
||||
const turnMetrics = useMemo(() => deriveTurnMetrics(nodes), [nodes])
|
||||
|
||||
@@ -631,8 +632,6 @@ export function ChatView({
|
||||
<MessageItem
|
||||
node={node}
|
||||
retryActive={node.kind === 'model-retry' && node.seq === activeRetry}
|
||||
onFork={forkAt}
|
||||
forkUnavailable={!branchSeqs.has(node.seq)}
|
||||
t={t}
|
||||
/>
|
||||
)
|
||||
|
||||
@@ -27,8 +27,6 @@ export interface MessageIconActionsProps {
|
||||
onBranch?: (() => void) | undefined
|
||||
/** The message is not a completed transcript tail, so branch stays visible but unavailable. */
|
||||
branchUnavailable?: boolean | undefined
|
||||
/** Additional branch visibility gate for transient message chrome; defaults to true. */
|
||||
showBranch?: boolean | undefined
|
||||
/** Parent layout class composed onto the actions row. */
|
||||
className?: string | undefined
|
||||
/** The owning view's locale seat, passed down as a plain prop. */
|
||||
@@ -41,7 +39,7 @@ export interface MessageIconActionsProps {
|
||||
* @returns The actions row element.
|
||||
*/
|
||||
export function MessageIconActions({
|
||||
text, time, runMs, ttftMs, tokensPerSecond, clock, onBranch, branchUnavailable = false, showBranch = true, className, t,
|
||||
text, time, runMs, ttftMs, tokensPerSecond, clock, onBranch, branchUnavailable = false, className, t,
|
||||
}: MessageIconActionsProps) {
|
||||
const day = useCalendarDay()
|
||||
const reasonId = useId()
|
||||
@@ -111,7 +109,7 @@ export function MessageIconActions({
|
||||
{copied ? <IconCheckOutline16 /> : <IconCopyOutline16 />}
|
||||
</button>
|
||||
</Tooltip>
|
||||
{showBranch && onBranch !== undefined && (
|
||||
{onBranch !== undefined && (
|
||||
<Tooltip label={branchUnavailable ? t('message.branchUnavailable') : t('message.branch')} side="bottom">
|
||||
{/* Native disabled buttons do not deliver the hover/focus events Tooltip needs. */}
|
||||
<button
|
||||
@@ -127,7 +125,7 @@ export function MessageIconActions({
|
||||
</button>
|
||||
</Tooltip>
|
||||
)}
|
||||
{showBranch && onBranch !== undefined && branchUnavailable && (
|
||||
{onBranch !== undefined && branchUnavailable && (
|
||||
<span id={reasonId} className={css.visuallyHidden}>{t('message.branchUnavailable')}</span>
|
||||
)}
|
||||
{clock === 'end' ? clockEl : null}
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
// MessageItem: simple chat nodes — user and consumed-steering bubbles
|
||||
// (right-aligned, with clock + copy / branch IconActions; steering adds the
|
||||
// interjection caption that names it), pending steering (caption + copy only),
|
||||
// context injection, compaction marker, retry disclosure, and unknown-surface
|
||||
// JSON rows.
|
||||
// (right-aligned, with clock + copy IconActions; steering adds the
|
||||
// interjection caption that names it; branch lives only under assistant
|
||||
// answers), pending steering (caption + copy only), context injection,
|
||||
// compaction marker, retry disclosure, and unknown-surface JSON rows.
|
||||
|
||||
import { memo, useEffect, useMemo, useState } from 'react'
|
||||
import type { ReactNode } from 'react'
|
||||
@@ -27,10 +27,6 @@ export interface MessageItemProps {
|
||||
| TurnErrorNode
|
||||
| UnknownSurfaceNode
|
||||
retryActive?: boolean
|
||||
/** Fork through this message's completed turn when eligible. */
|
||||
onFork?: (seq: number) => void
|
||||
/** The message is not the transcript tail of a completed turn. */
|
||||
forkUnavailable?: boolean
|
||||
/** The owning view's locale seat, passed down as a plain prop. */
|
||||
t: ChatViewSlotProps['t']
|
||||
}
|
||||
@@ -217,7 +213,6 @@ export function PendingSteeringBubble({ content, t }: {
|
||||
<MessageIconActions
|
||||
text={text}
|
||||
clock="start"
|
||||
showBranch={false}
|
||||
className={css.actions}
|
||||
t={t}
|
||||
/>
|
||||
@@ -227,7 +222,7 @@ export function PendingSteeringBubble({ content, t }: {
|
||||
}
|
||||
|
||||
export const MessageItem = memo(function MessageItem({
|
||||
node, retryActive = false, onFork, forkUnavailable = false, t,
|
||||
node, retryActive = false, t,
|
||||
}: MessageItemProps) {
|
||||
const truncated = (total: number): string => t('json.truncated', { total })
|
||||
switch (node.kind) {
|
||||
@@ -243,8 +238,6 @@ export const MessageItem = memo(function MessageItem({
|
||||
text={text}
|
||||
time={node.time}
|
||||
clock="start"
|
||||
onBranch={onFork === undefined ? undefined : () => { onFork(node.seq) }}
|
||||
branchUnavailable={forkUnavailable}
|
||||
className={css.actions}
|
||||
t={t}
|
||||
/>
|
||||
|
||||
@@ -89,6 +89,20 @@
|
||||
text-overflow: clip;
|
||||
}
|
||||
|
||||
/* Trailing summary fragment kept out of .summary's ellipsis, for a count whose
|
||||
whole value is that it survives a narrow row (the todo row's parallel-active
|
||||
`+n`). Repeats .summary's type because it sits beside that text, and its
|
||||
`nowrap` too: `flex: none` stops the box shrinking but not the text wrapping,
|
||||
which would break the one-line row in the narrow case the slot exists for. */
|
||||
.summarySuffix {
|
||||
flex: none;
|
||||
margin-left: 4px;
|
||||
white-space: nowrap;
|
||||
font-size: 14px;
|
||||
line-height: 24px;
|
||||
color: var(--dsw-alias-label-tertiary);
|
||||
}
|
||||
|
||||
/* File-tool path: same geometry as .summary; hover underline + pointer. */
|
||||
.fileLink {
|
||||
flex: 1 1 auto;
|
||||
|
||||
@@ -46,6 +46,14 @@ export interface ToolRowProps {
|
||||
icon: ReactNode
|
||||
title: string
|
||||
summary: string
|
||||
/**
|
||||
* Trailing summary fragment rendered outside the ellipsized summary text, so
|
||||
* a narrow row clips the summary before this. For a fragment whose whole
|
||||
* value is surviving that clip — the todo row's parallel-active count.
|
||||
* null/absent = the summary is the whole collapsed content. Dropped on an
|
||||
* error row, whose collapsed summary is the failure line instead.
|
||||
*/
|
||||
summarySuffix?: string | null | undefined
|
||||
/** Expanded-body input text; null = no input section. */
|
||||
body: string | null
|
||||
/** Flattened result text for the expanded Output section; null/absent = no output section. */
|
||||
@@ -139,6 +147,7 @@ export function ToolRow({
|
||||
icon,
|
||||
title,
|
||||
summary,
|
||||
summarySuffix,
|
||||
body,
|
||||
output,
|
||||
errorSummary,
|
||||
@@ -173,6 +182,9 @@ export function ToolRow({
|
||||
// the error color outranks both the args summary and a terminal description.
|
||||
const failureLine = state === 'error' ? errorSummary ?? null : null
|
||||
const summaryText = failureLine ?? summary
|
||||
// The failure line replaces the summary wholesale, so a suffix derived from
|
||||
// the call args has nothing left to sit beside.
|
||||
const suffix = failureLine === null ? summarySuffix ?? null : null
|
||||
// The failure line is error prose, not the path: no open-file affordance.
|
||||
const fileLink = filePath !== undefined && onOpenFile !== undefined && failureLine === null
|
||||
const isThink = variant === 'think'
|
||||
@@ -249,6 +261,7 @@ export function ToolRow({
|
||||
{summaryText}
|
||||
</span>
|
||||
)}
|
||||
{suffix !== null && <span className={css.summarySuffix}>{suffix}</span>}
|
||||
</>
|
||||
)}
|
||||
>
|
||||
|
||||
@@ -17,8 +17,14 @@ export type ChatFlowItem =
|
||||
| { kind: 'node'; key: string; node: ConversationNode }
|
||||
| { kind: 'tool-group'; key: string; results: readonly ToolResultNode[] }
|
||||
|
||||
/** True when the node has model-visible text content worth IconActions chrome. */
|
||||
function hasContentText(blocks: readonly AssistantBlock[]): boolean {
|
||||
/**
|
||||
* True when the node has model-visible text content worth IconActions chrome.
|
||||
* Shared with {@link AssistantMarkdown}'s mount gate so ownership and mounting
|
||||
* cannot diverge.
|
||||
* @param blocks - assistant blocks of one finalized node.
|
||||
* @returns Whether any text block carries non-blank content.
|
||||
*/
|
||||
export function hasContentText(blocks: readonly AssistantBlock[]): boolean {
|
||||
return blocks.some(block => block.kind === 'text' && block.text.trim() !== '')
|
||||
}
|
||||
|
||||
@@ -34,14 +40,20 @@ function rendersNothing(node: ConversationNode): boolean {
|
||||
|
||||
/**
|
||||
* Seq set of assistants that own IconActions: the last content-text assistant
|
||||
* in each turn. Mid-turn narration (text before tools) stays chrome-free.
|
||||
* of each *completed* turn. A turn without a `turn/end` in the window is still
|
||||
* producing steps, so its latest narration is not the settled answer and owns
|
||||
* nothing; mid-turn narration of a completed turn stays chrome-free too.
|
||||
* @param nodes - snapshot nodes (surface order).
|
||||
* @param turnEnds - completed turn boundaries retained from the event window.
|
||||
* @returns Seq values ChatView may pass as `time` into AssistantMarkdown.
|
||||
*/
|
||||
export function assistantActionsSeqs(nodes: readonly ConversationNode[]): ReadonlySet<number> {
|
||||
export function assistantActionsSeqs(
|
||||
nodes: readonly ConversationNode[],
|
||||
turnEnds: ReadonlyMap<number, number>,
|
||||
): ReadonlySet<number> {
|
||||
const lastByTurn = new Map<number, number>()
|
||||
for (const node of nodes) {
|
||||
if (node.kind !== 'assistant' || !hasContentText(node.blocks)) continue
|
||||
if (node.kind !== 'assistant' || !turnEnds.has(node.turn) || !hasContentText(node.blocks)) continue
|
||||
lastByTurn.set(node.turn, node.seq)
|
||||
}
|
||||
return new Set(lastByTurn.values())
|
||||
@@ -63,15 +75,18 @@ export function runningTurnStartTime(
|
||||
}
|
||||
|
||||
/**
|
||||
* Seq set of message rows that may fork: the last transcript node of a
|
||||
* completed turn, when that node owns message chrome. A later tool, reasoning,
|
||||
* error, or other transcript node leaves the earlier message's branch action
|
||||
* unavailable because the Host would include the whole turn.
|
||||
* Seq set of assistant answers that may fork: the completed turn's transcript
|
||||
* tail, when that tail is the turn's own content-text assistant. A later tool,
|
||||
* reasoning, error, or other transcript node leaves the answer's branch action
|
||||
* unavailable because the Host would include the whole turn. User and steering
|
||||
* bubbles carry no branch action at all: a fork at their seq cuts at the same
|
||||
* `turn/end` as the answer's, so the affordance lives only under the settled
|
||||
* answer.
|
||||
* @param nodes - snapshot nodes in event order.
|
||||
* @param turnEnds - completed turn boundaries retained from the event window.
|
||||
* @returns Message seq values whose visible position matches the fork boundary.
|
||||
* @returns Assistant seq values whose visible position matches the fork boundary.
|
||||
*/
|
||||
export function messageBranchSeqs(
|
||||
export function assistantBranchSeqs(
|
||||
nodes: readonly ConversationNode[],
|
||||
turnEnds: ReadonlyMap<number, number>,
|
||||
): ReadonlySet<number> {
|
||||
@@ -86,8 +101,7 @@ export function messageBranchSeqs(
|
||||
tail = candidate
|
||||
nodeIndex++
|
||||
}
|
||||
if (tail?.kind === 'user' || tail?.kind === 'steering'
|
||||
|| (tail?.kind === 'assistant' && tail.turn === turn && hasContentText(tail.blocks))) {
|
||||
if (tail?.kind === 'assistant' && tail.turn === turn && hasContentText(tail.blocks)) {
|
||||
result.add(tail.seq)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
/**
|
||||
* Pure plan derivation for the todo_write row's one-line summary. Several items
|
||||
* may be `in_progress` at once — parallel work runs concurrent tasks, so a
|
||||
* summary built from one active item would silently drop the rest. The plan
|
||||
* strip header derives its own counts inline and shares nothing with this, so
|
||||
* this stays inside the toolviews domain rather than in `contract/` (the
|
||||
* inter-domain face).
|
||||
* @module
|
||||
*/
|
||||
|
||||
/**
|
||||
* One list item as the row sees it: unvalidated model JSON parsed from a call's
|
||||
* args, so any field may be missing or mistyped.
|
||||
*/
|
||||
export interface PlanItemLike {
|
||||
content?: unknown
|
||||
status?: unknown
|
||||
}
|
||||
|
||||
/**
|
||||
* Counts plus the two halves of the summary, deliberately NOT pre-joined: the
|
||||
* row ellipsizes its summary text, and a count concatenated onto the end of the
|
||||
* task name is the first thing a narrow row clips — exactly when it carries
|
||||
* information. The row renders `activeExtra` in its own non-shrinking span
|
||||
* beside the truncatable text.
|
||||
*/
|
||||
export interface PlanSummary {
|
||||
done: number
|
||||
total: number
|
||||
/** First `in_progress` content, or null when that first item is unusable. */
|
||||
activeContent: string | null
|
||||
/** Active items beyond the first; 0 whenever there is no `activeContent` to sit beside. */
|
||||
activeExtra: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Derive the counts and the active summary from a whole-list snapshot. It names
|
||||
* the first `in_progress` item and counts the remaining active ones, so a
|
||||
* parallel plan reports how many tasks are running rather than naming one and
|
||||
* hiding the others. `activeContent` is null when nothing is in progress, or
|
||||
* when the first active item's content is missing, mistyped, or blank once
|
||||
* trimmed — the tool's own rule for usable content, applied here because a
|
||||
* rejected call keeps its args verbatim. The row then renders the counts alone
|
||||
* rather than falling back to the generic tool summary: the counts are already
|
||||
* known to be good, and the active-item clause is the only part an unusable
|
||||
* name costs.
|
||||
* @param todos - the whole list, in model order.
|
||||
* @returns the done/total counts and the two summary halves.
|
||||
*/
|
||||
export function planSummary(todos: readonly PlanItemLike[]): PlanSummary {
|
||||
const active = todos.filter(t => t.status === 'in_progress')
|
||||
const first = active[0]?.content
|
||||
const named = typeof first === 'string' && first.trim() !== ''
|
||||
return {
|
||||
done: todos.filter(t => t.status === 'completed').length,
|
||||
total: todos.length,
|
||||
activeContent: named ? first : null,
|
||||
activeExtra: named ? active.length - 1 : 0,
|
||||
}
|
||||
}
|
||||
@@ -2,9 +2,10 @@
|
||||
// "Tool call" card, registered into the keyed 'conversation.chat.toolview'
|
||||
// hole like the bash sample (a product registration, not a sample). The row
|
||||
// composes ToolRow (chrome, running sweep, whole-row expand) and swaps in a
|
||||
// summary of the written list (counts + active item) from the call args; the
|
||||
// durable list itself renders in the TodoPanel above the composer, so the
|
||||
// row stays one line until expanded.
|
||||
// summary of the written list (counts + active items) from the call args, with
|
||||
// the parallel-active count riding ToolRow's non-shrinking summary suffix so a
|
||||
// narrow row never clips it; the durable list itself renders in the TodoPanel
|
||||
// above the composer, so the row stays one line until expanded.
|
||||
|
||||
import { IconChecklistOutline14 } from '@deepseek-ai/dsh-client-ui-primitives'
|
||||
import type { Context } from 'cordis'
|
||||
@@ -13,18 +14,26 @@ import type { ToolRowProps } from '../contract/slots.ts'
|
||||
import { toolRowModel } from '../contract/tool-call-model.ts'
|
||||
import { ToolRow } from '../chat/ToolRow.tsx'
|
||||
import { NS } from '../locales.ts'
|
||||
import { planSummary, type PlanItemLike } from './plan-summary.ts'
|
||||
|
||||
/** Todo row props: the toolview runtime share plus the standard locale seat. */
|
||||
type TodoRowProps = ToolRowProps & PropsLocale<'conversation'>
|
||||
|
||||
/** One parsed args item, shape-checked (model JSON: any field may be missing or mistyped). */
|
||||
interface TodoWriteItem { content?: unknown; status?: unknown }
|
||||
|
||||
function isItem(value: unknown): value is TodoWriteItem {
|
||||
function isItem(value: unknown): value is PlanItemLike {
|
||||
return typeof value === 'object' && value !== null
|
||||
}
|
||||
|
||||
function summarize(argsRaw: string, t: TodoRowProps['t']): string | null {
|
||||
/**
|
||||
* The row's summary split at the ellipsis boundary: `text` truncates, `extra`
|
||||
* is the parallel-active count that must not, so a narrow row never clips the
|
||||
* one part that says several tasks are running.
|
||||
*/
|
||||
interface RowSummary {
|
||||
text: string
|
||||
extra: number
|
||||
}
|
||||
|
||||
function summarize(argsRaw: string, t: TodoRowProps['t']): RowSummary | null {
|
||||
let parsed: unknown
|
||||
try {
|
||||
parsed = JSON.parse(argsRaw)
|
||||
@@ -37,12 +46,12 @@ function summarize(argsRaw: string, t: TodoRowProps['t']): string | null {
|
||||
if (typeof parsed !== 'object' || parsed === null) return null
|
||||
const todos = (parsed as { todos?: unknown }).todos
|
||||
if (!Array.isArray(todos) || !todos.every(isItem)) return null
|
||||
const done = todos.filter(item => item.status === 'completed').length
|
||||
const active = todos.find(item => item.status === 'in_progress')
|
||||
const head = t('todo.completed', { done, total: todos.length })
|
||||
return typeof active?.content === 'string' && active.content !== ''
|
||||
? `${head} · ${active.content}`
|
||||
: head
|
||||
const { done, total, activeContent, activeExtra } = planSummary(todos)
|
||||
const head = t('todo.completed', { done, total })
|
||||
return {
|
||||
text: activeContent === null ? head : `${head} · ${activeContent}`,
|
||||
extra: activeExtra,
|
||||
}
|
||||
}
|
||||
|
||||
/** One-line plan update row (the whole row toggles the call's Input/Output
|
||||
@@ -52,7 +61,7 @@ function summarize(argsRaw: string, t: TodoRowProps['t']): string | null {
|
||||
export function TodoRow({ toolName, block, inspect, t }: TodoRowProps) {
|
||||
const model = toolRowModel(toolName, block)
|
||||
const argsRaw = ('kind' in block ? block.call?.argsRaw : block.argsRaw) ?? ''
|
||||
const summary = summarize(argsRaw, t) ?? model.summary
|
||||
const summary = summarize(argsRaw, t) ?? { text: model.summary, extra: 0 }
|
||||
return (
|
||||
<ToolRow
|
||||
t={t}
|
||||
@@ -60,7 +69,8 @@ export function TodoRow({ toolName, block, inspect, t }: TodoRowProps) {
|
||||
toolName={toolName}
|
||||
icon={<IconChecklistOutline14 />}
|
||||
title={t('todo.rowTitle')}
|
||||
summary={summary}
|
||||
summary={summary.text}
|
||||
summarySuffix={summary.extra > 0 ? `+${summary.extra}` : null}
|
||||
body={model.body}
|
||||
output={model.output}
|
||||
errorSummary={model.errorSummary}
|
||||
|
||||
@@ -36,7 +36,7 @@ afterEach(() => {
|
||||
const t: MessageItemProps['t'] = makeTranslate(zh, commonZh)
|
||||
|
||||
describe('MessageItem arms', () => {
|
||||
it('user bubbles expose clock / copy / branch and no edit; copy writes the text', () => {
|
||||
it('user bubbles expose clock / copy and neither branch nor edit; copy writes the text', () => {
|
||||
const writeText = vi.fn().mockResolvedValue(undefined)
|
||||
Object.defineProperty(navigator, 'clipboard', {
|
||||
configurable: true,
|
||||
@@ -45,24 +45,20 @@ describe('MessageItem arms', () => {
|
||||
// Same-day clock: construct "today at 14:24" so the label stays `HH:mm`.
|
||||
const now = new Date()
|
||||
const time = new Date(now.getFullYear(), now.getMonth(), now.getDate(), 14, 24).getTime()
|
||||
const onFork = vi.fn()
|
||||
render(
|
||||
<MessageItem t={t} node={{
|
||||
kind: 'user', seq: 1, time,
|
||||
content: [{ type: 'text', text: 'hello bubble' }] as never,
|
||||
source: null,
|
||||
}}
|
||||
onFork={onFork}
|
||||
/>,
|
||||
)
|
||||
expect(screen.getByText('14:24')).toBeTruthy()
|
||||
expect(screen.getByRole('button', { name: '复制' })).toBeTruthy()
|
||||
expect(screen.getByRole('button', { name: '在新对话中分支' })).toBeTruthy()
|
||||
expect(screen.queryByRole('button', { name: '在新对话中分支' })).toBeNull()
|
||||
expect(screen.queryByRole('button', { name: '编辑' })).toBeNull()
|
||||
fireEvent.click(screen.getByRole('button', { name: '复制' }))
|
||||
expect(writeText).toHaveBeenCalledWith('hello bubble')
|
||||
fireEvent.click(screen.getByRole('button', { name: '在新对话中分支' }))
|
||||
expect(onFork).toHaveBeenCalledWith(1)
|
||||
})
|
||||
|
||||
it('user copy falls back to execCommand when clipboard.writeText is unavailable', () => {
|
||||
@@ -87,30 +83,6 @@ describe('MessageItem arms', () => {
|
||||
expect(exec).toHaveBeenCalledWith('copy')
|
||||
})
|
||||
|
||||
it('keeps an unavailable branch focusable and explains why without sending a fork', () => {
|
||||
const onFork = vi.fn()
|
||||
render(
|
||||
<MessageItem t={t} node={{
|
||||
kind: 'user', seq: 1, time: 1_000,
|
||||
content: [{ type: 'text', text: 'open turn' }] as never,
|
||||
source: null,
|
||||
}}
|
||||
onFork={onFork}
|
||||
forkUnavailable
|
||||
/>,
|
||||
)
|
||||
const branch = screen.getByRole('button', { name: '在新对话中分支' }) as HTMLButtonElement
|
||||
expect(branch.disabled).toBe(false)
|
||||
expect(branch.getAttribute('aria-disabled')).toBe('true')
|
||||
const reasonId = branch.getAttribute('aria-describedby')
|
||||
expect(reasonId).not.toBeNull()
|
||||
expect(document.getElementById(reasonId!)?.textContent).toBe('仅可从已完成轮次的最后一条消息分支')
|
||||
fireEvent.click(branch)
|
||||
expect(onFork).not.toHaveBeenCalled()
|
||||
fireEvent.focus(branch)
|
||||
expect(screen.getByRole('tooltip').textContent).toBe('仅可从已完成轮次的最后一条消息分支')
|
||||
})
|
||||
|
||||
it('user copy never claims success when the host rejects the write', async () => {
|
||||
Object.defineProperty(navigator, 'clipboard', {
|
||||
configurable: true,
|
||||
@@ -212,19 +184,17 @@ describe('MessageItem arms', () => {
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
})
|
||||
|
||||
it('consumed steering is captioned as an interjection and keeps copy and branch actions', () => {
|
||||
it('consumed steering is captioned as an interjection and keeps copy without branch', () => {
|
||||
const writeText = vi.fn().mockResolvedValue(undefined)
|
||||
Object.defineProperty(navigator, 'clipboard', {
|
||||
configurable: true,
|
||||
value: { writeText },
|
||||
})
|
||||
const fork = vi.fn()
|
||||
const view = render(
|
||||
<MessageItem t={t} node={{
|
||||
kind: 'steering', messageId: 'steer-message', seq: 2, time: 1_000, turn: 1, source: null,
|
||||
content: [{ type: 'text', text: 'steer!' }, { type: 'image', data: 'x' }] as never,
|
||||
} as never}
|
||||
onFork={fork}
|
||||
/>,
|
||||
)
|
||||
expect(view.getByText('插话')).toBeTruthy()
|
||||
@@ -232,8 +202,7 @@ describe('MessageItem arms', () => {
|
||||
expect(view.getByText(/附加内容块/)).toBeTruthy()
|
||||
fireEvent.click(view.getByRole('button', { name: '复制' }))
|
||||
expect(writeText).toHaveBeenCalledWith('steer!')
|
||||
fireEvent.click(view.getByRole('button', { name: '在新对话中分支' }))
|
||||
expect(fork).toHaveBeenCalledWith(2)
|
||||
expect(view.queryByRole('button', { name: '在新对话中分支' })).toBeNull()
|
||||
})
|
||||
|
||||
it('context uses the Tool calls disclosure chrome and keeps its body collapsed by default', () => {
|
||||
@@ -1002,6 +971,31 @@ describe('small branch tails', () => {
|
||||
expect(streaming.queryByText('14:24')).toBeNull()
|
||||
})
|
||||
|
||||
it('keeps an unavailable branch focusable and explains why without sending a fork', () => {
|
||||
const onFork = vi.fn()
|
||||
render(
|
||||
<AssistantMarkdown
|
||||
t={t}
|
||||
blocks={[{ kind: 'text', text: 'answer before a trailing tool row' }]}
|
||||
streaming={false}
|
||||
time={1_000}
|
||||
seq={1}
|
||||
onFork={onFork}
|
||||
forkUnavailable
|
||||
/>,
|
||||
)
|
||||
const branch = screen.getByRole('button', { name: '在新对话中分支' }) as HTMLButtonElement
|
||||
expect(branch.disabled).toBe(false)
|
||||
expect(branch.getAttribute('aria-disabled')).toBe('true')
|
||||
const reasonId = branch.getAttribute('aria-describedby')
|
||||
expect(reasonId).not.toBeNull()
|
||||
expect(document.getElementById(reasonId!)?.textContent).toBe('仅可从已完成轮次的最后一条消息分支')
|
||||
fireEvent.click(branch)
|
||||
expect(onFork).not.toHaveBeenCalled()
|
||||
fireEvent.focus(branch)
|
||||
expect(screen.getByRole('tooltip').textContent).toBe('仅可从已完成轮次的最后一条消息分支')
|
||||
})
|
||||
|
||||
it('StatsLine omits the cache-hit segment when no input accounting exists at all', () => {
|
||||
// Cache hit is null only when all three prompt buckets are zero (pure
|
||||
// output accounting) — any billed input makes it a real 0%.
|
||||
|
||||
@@ -301,6 +301,20 @@ describe('ToolRow', () => {
|
||||
expect(view.getByText('List files')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('renders summarySuffix outside the ellipsized summary span, and drops it on a failure line', () => {
|
||||
const view = render(<ToolRow {...rowProps} summarySuffix="+2" />)
|
||||
const summary = view.getByText('List files')
|
||||
const suffix = view.getByText('+2')
|
||||
// Separate spans: .summary truncates, the suffix must not travel inside it.
|
||||
expect(summary.contains(suffix)).toBe(false)
|
||||
view.unmount()
|
||||
// The failure line replaces the summary wholesale, so the suffix goes with it.
|
||||
const failed = render(
|
||||
<ToolRow {...rowProps} state="error" errorSummary="boom" summarySuffix="+2" />,
|
||||
)
|
||||
expect(failed.queryByText('+2')).toBeNull()
|
||||
})
|
||||
|
||||
it('an error file row drops the open-file link (the summary is failure prose, not the path)', () => {
|
||||
const open = vi.fn()
|
||||
const view = render(
|
||||
|
||||
@@ -20,7 +20,7 @@ import { zh as commonZh } from '@deepseek-ai/dsh-client-locale/src/locales/zh.ts
|
||||
import { createChatStore } from '../src/client/stores.ts'
|
||||
import { ChatView } from '../src/client/chat/ChatView.tsx'
|
||||
import { zh } from '../src/client/locales.ts'
|
||||
import { assistantActionsSeqs, deriveChatFlow, flowKeys, messageBranchSeqs, runningTurnStartTime } from '../src/client/chat/chat-flow.ts'
|
||||
import { assistantActionsSeqs, assistantBranchSeqs, deriveChatFlow, flowKeys, runningTurnStartTime } from '../src/client/chat/chat-flow.ts'
|
||||
import { formatRunDuration } from '../src/client/chat/message-chrome.ts'
|
||||
|
||||
afterEach(() => {
|
||||
@@ -225,12 +225,12 @@ describe('chat-flow derivation', () => {
|
||||
expect(flowKeys(deriveChatFlow([toolResult(3, 'a'), assistant(4, 'found'), toolResult(5, 'b')]))).toBe('g3|n4|g5')
|
||||
})
|
||||
|
||||
it('assistantActionsSeqs keeps only the last content assistant per turn', () => {
|
||||
it('assistantActionsSeqs keeps only the last content assistant per completed turn', () => {
|
||||
const thinkOnly: AssistantMessageNode = {
|
||||
kind: 'assistant', seq: 3, time: 3_000, turn: 1, step: 2,
|
||||
blocks: [{ kind: 'reasoning', text: 'planning' }],
|
||||
}
|
||||
const seqs = assistantActionsSeqs([
|
||||
const nodes: ConversationNode[] = [
|
||||
user(1, 'hi'),
|
||||
assistant(2, 'looking', 1),
|
||||
thinkOnly,
|
||||
@@ -238,8 +238,11 @@ describe('chat-flow derivation', () => {
|
||||
assistant(5, 'done', 1),
|
||||
user(6, 'again'),
|
||||
assistant(7, 'second turn', 2),
|
||||
])
|
||||
expect([...seqs].sort((a, b) => a - b)).toEqual([5, 7])
|
||||
]
|
||||
expect([...assistantActionsSeqs(nodes, new Map([[1, 5], [2, 7]]))].sort((a, b) => a - b)).toEqual([5, 7])
|
||||
// Turn 2 is still producing steps: its latest narration owns nothing, and
|
||||
// the settled turn 1 keeps its seat.
|
||||
expect([...assistantActionsSeqs(nodes, new Map([[1, 5]]))]).toEqual([5])
|
||||
})
|
||||
|
||||
it('runningTurnStartTime selects the latest turn/start without a turn/end', () => {
|
||||
@@ -261,7 +264,7 @@ describe('chat-flow derivation', () => {
|
||||
expect(formatRunDuration(125_000, t)).toBe('2分05秒')
|
||||
})
|
||||
|
||||
it('messageBranchSeqs keeps only message rows at completed transcript tails', () => {
|
||||
it('assistantBranchSeqs keeps only content-assistant tails; user/steering tails own no branch', () => {
|
||||
const interruptedThink: AssistantMessageNode = {
|
||||
kind: 'assistant', seq: 4.1, time: 4_100, turn: 1, step: 2,
|
||||
blocks: [{ kind: 'reasoning', text: 'bad path' }], interrupted: true,
|
||||
@@ -276,8 +279,8 @@ describe('chat-flow derivation', () => {
|
||||
user(10, 'user-only tail'),
|
||||
user(13, 'steering tail'),
|
||||
]
|
||||
const seqs = messageBranchSeqs(nodes, new Map([[1, 5], [2, 8], [3, 11], [4, 14]]))
|
||||
expect([...seqs]).toEqual([7, 10, 13])
|
||||
const seqs = assistantBranchSeqs(nodes, new Map([[1, 5], [2, 8], [3, 11], [4, 14]]))
|
||||
expect([...seqs]).toEqual([7])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -401,21 +404,24 @@ describe('ChatView', () => {
|
||||
expect(view.getAllByText('interrupt now')).toHaveLength(1)
|
||||
expect(view.container.querySelector('[data-pending-steering]')).toBeNull()
|
||||
expect(view.getAllByText('插话')).toHaveLength(1)
|
||||
expect(view.getAllByRole('button', { name: '复制' })).toHaveLength(2)
|
||||
// Only the durable steering bubble: the turn is still running, so its
|
||||
// assistant narration owns no footer yet, and a steering bubble never
|
||||
// carries a branch action.
|
||||
expect(view.getAllByRole('button', { name: '复制' })).toHaveLength(1)
|
||||
const durableBubble = view.getByText('interrupt now').closest('[class*="userRow"]') as HTMLElement
|
||||
const unavailable = within(durableBubble).getByRole('button', { name: '在新对话中分支' })
|
||||
expect(unavailable.getAttribute('aria-disabled')).toBe('true')
|
||||
fireEvent.click(unavailable)
|
||||
expect(h.forkAt).not.toHaveBeenCalled()
|
||||
expect(within(durableBubble).queryByRole('button', { name: '在新对话中分支' })).toBeNull()
|
||||
|
||||
act(() => {
|
||||
h.set({ running: false, turnEnds: new Map([[1, 3]]) })
|
||||
})
|
||||
// The completed turn's transcript tail is the steering bubble, not the
|
||||
// narration, so the assistant's branch action stays unavailable and the
|
||||
// steering bubble still offers none.
|
||||
const branchButtons = view.getAllByRole('button', { name: '在新对话中分支' })
|
||||
expect(branchButtons).toHaveLength(2)
|
||||
expect(branchButtons.map(button => button.getAttribute('aria-disabled'))).toEqual(['true', null])
|
||||
fireEvent.click(branchButtons[1]!)
|
||||
expect(h.forkAt).toHaveBeenCalledWith(2)
|
||||
expect(branchButtons).toHaveLength(1)
|
||||
expect(branchButtons[0]!.getAttribute('aria-disabled')).toBe('true')
|
||||
fireEvent.click(branchButtons[0]!)
|
||||
expect(h.forkAt).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
it('keeps a later pending occurrence visible when it reuses a durable MessageId', () => {
|
||||
@@ -518,11 +524,35 @@ describe('ChatView', () => {
|
||||
turnEnds: new Map([[1, 4], [2, 6]]),
|
||||
})
|
||||
const view = render(<h.ChatView {...h.props} />)
|
||||
// Every message footer keeps branch visible; only completed assistant tails enable it.
|
||||
// Branch renders only under assistant answers; user bubbles keep copy alone.
|
||||
expect(view.getAllByRole('button', { name: '复制' })).toHaveLength(4)
|
||||
const branchButtons = view.getAllByRole('button', { name: '在新对话中分支' })
|
||||
expect(branchButtons).toHaveLength(4)
|
||||
expect(branchButtons.map(button => button.getAttribute('aria-disabled'))).toEqual(['true', null, 'true', null])
|
||||
expect(branchButtons).toHaveLength(2)
|
||||
expect(branchButtons.map(button => button.getAttribute('aria-disabled'))).toEqual([null, null])
|
||||
})
|
||||
|
||||
it('withholds assistant IconActions while the turn is still running', () => {
|
||||
const h = makeHarness({
|
||||
running: true,
|
||||
runningCalls: [runningCall('a')],
|
||||
nodes: [
|
||||
user(1, 'first'),
|
||||
assistant(2, 'previous answer', 1),
|
||||
user(4, 'second'),
|
||||
assistant(5, 'mid-turn text', 2),
|
||||
],
|
||||
// Boundary seqs follow the log: a turn/end is strictly after its own nodes.
|
||||
turnEnds: new Map([[1, 3]]),
|
||||
})
|
||||
const view = render(<h.ChatView {...h.props} />)
|
||||
// 2 user + the settled turn-1 tail, which keeps its seat while a later
|
||||
// turn runs; turn 2's narration stays chrome-free while its tool runs, so
|
||||
// the footer never appears and then moves.
|
||||
expect(view.getAllByRole('button', { name: '复制' })).toHaveLength(3)
|
||||
expect(view.getByText('mid-turn text')).toBeTruthy()
|
||||
// turn/end lands: the same node becomes the settled answer and takes the seat.
|
||||
act(() => { h.set({ running: false, runningCalls: [], turnEnds: new Map([[1, 3], [2, 6]]) }) })
|
||||
expect(view.getAllByRole('button', { name: '复制' })).toHaveLength(4)
|
||||
})
|
||||
|
||||
it('the actions-owning assistant footer shows the turn run time', () => {
|
||||
@@ -606,11 +636,11 @@ describe('ChatView', () => {
|
||||
turnEnds: new Map([[1, 3]]),
|
||||
})
|
||||
const view = render(<h.ChatView {...h.props} />)
|
||||
// The user bubble offers no branch; the settled answer's is live.
|
||||
const buttons = view.getAllByRole('button', { name: '在新对话中分支' })
|
||||
expect(buttons).toHaveLength(2)
|
||||
expect(buttons.map(button => button.getAttribute('aria-disabled'))).toEqual(['true', null])
|
||||
expect(buttons).toHaveLength(1)
|
||||
expect(buttons[0]!.getAttribute('aria-disabled')).toBeNull()
|
||||
fireEvent.click(buttons[0]!)
|
||||
fireEvent.click(buttons[1]!)
|
||||
expect(h.forkAt.mock.calls).toEqual([[2]])
|
||||
})
|
||||
|
||||
@@ -626,10 +656,9 @@ describe('ChatView', () => {
|
||||
const view = render(<h.ChatView {...h.props} />)
|
||||
expect(view.getAllByRole('button', { name: '复制' })).toHaveLength(2)
|
||||
const buttons = view.getAllByRole('button', { name: '在新对话中分支' })
|
||||
expect(buttons).toHaveLength(2)
|
||||
expect(buttons.every(button => button.getAttribute('aria-disabled') === 'true')).toBe(true)
|
||||
expect(buttons).toHaveLength(1)
|
||||
expect(buttons[0]!.getAttribute('aria-disabled')).toBe('true')
|
||||
fireEvent.click(buttons[0]!)
|
||||
fireEvent.click(buttons[1]!)
|
||||
expect(h.forkAt).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
// @vitest-environment jsdom
|
||||
/**
|
||||
* Todo display acceptance: the TodoPanel plan strip (empty-hidden, status
|
||||
* rows, collapse), its TodoDock adapter (selects the plan off the session
|
||||
* snapshot and follows changes), and the todo_write toolview row (progress
|
||||
* summary from args, generic fallback on malformed JSON, shared ToolRow
|
||||
* state dots and leading expansion).
|
||||
* Todo display acceptance: the TodoPanel plan strip (empty-hidden, status rows
|
||||
* including several `in_progress` at once, collapse), its TodoDock adapter
|
||||
* (selects the plan off the session snapshot and follows changes), the row's
|
||||
* plan summary (counts plus the two halves of the active summary — the named
|
||||
* task and the `+N` count that parallel work adds, kept apart so the row never
|
||||
* ellipsizes the count away), and the todo_write toolview row (progress summary
|
||||
* from args, generic fallback on malformed JSON, shared ToolRow state dots and
|
||||
* leading expansion).
|
||||
*/
|
||||
import { act, cleanup, fireEvent, render, screen } from '@testing-library/react'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
@@ -17,6 +20,7 @@ import { zh as commonZh } from '@deepseek-ai/dsh-client-locale/src/locales/zh.ts
|
||||
import { TodoRow, todoToolview } from '../src/client/toolviews/todo-row.tsx'
|
||||
import type { TodoDockProps } from '../src/client/skeleton/TodoPanel.tsx'
|
||||
import { TodoDock, TodoPanel, todoDockEntry } from '../src/client/skeleton/TodoPanel.tsx'
|
||||
import { planSummary } from '../src/client/toolviews/plan-summary.ts'
|
||||
import { NS, zh } from '../src/client/locales.ts'
|
||||
|
||||
type TodoRowProps = Parameters<typeof TodoRow>[0]
|
||||
@@ -32,6 +36,49 @@ const LIST: TodoItem[] = [
|
||||
{ content: '补测试', status: 'pending' },
|
||||
]
|
||||
|
||||
/** A parallel plan: three tasks running at once (concurrent subagents). */
|
||||
const PARALLEL: TodoItem[] = [
|
||||
{ content: '搭骨架', status: 'completed' },
|
||||
{ content: '写组件', status: 'in_progress' },
|
||||
{ content: '跑后台构建', status: 'in_progress' },
|
||||
{ content: '读源码', status: 'in_progress' },
|
||||
{ content: '补测试', status: 'pending' },
|
||||
]
|
||||
|
||||
describe('planSummary', () => {
|
||||
it('counts done/total and names the single active item with no extra count', () => {
|
||||
expect(planSummary(LIST)).toEqual({ done: 1, total: 3, activeContent: '写组件', activeExtra: 0 })
|
||||
})
|
||||
|
||||
it('reports the extra active count separately when several items are in progress', () => {
|
||||
// Parallel work marks several: naming one and hiding the rest would lose
|
||||
// them, and the count stays unjoined so the row cannot ellipsize it.
|
||||
expect(planSummary(PARALLEL)).toEqual({ done: 1, total: 5, activeContent: '写组件', activeExtra: 2 })
|
||||
})
|
||||
|
||||
it('has no hint when nothing is in progress', () => {
|
||||
expect(planSummary([{ content: '都完了', status: 'completed' }]))
|
||||
.toEqual({ done: 1, total: 1, activeContent: null, activeExtra: 0 })
|
||||
})
|
||||
|
||||
it('has no hint when the first active item carries no usable content (model JSON)', () => {
|
||||
// Unvalidated args: a missing, mistyped, empty, or whitespace-only content
|
||||
// yields no hint — and no orphan count, even with a second active item to
|
||||
// count. Whitespace-only is the tool's own rejection rule (trimmed
|
||||
// non-empty), and a rejected call keeps its args verbatim.
|
||||
expect(planSummary([{ status: 'in_progress' }, { content: 'x', status: 'in_progress' }]))
|
||||
.toMatchObject({ activeContent: null, activeExtra: 0 })
|
||||
expect(planSummary([{ content: 42, status: 'in_progress' }]).activeContent).toBeNull()
|
||||
expect(planSummary([{ content: '', status: 'in_progress' }]).activeContent).toBeNull()
|
||||
expect(planSummary([{ content: ' ', status: 'in_progress' }, { content: 'x', status: 'in_progress' }]))
|
||||
.toMatchObject({ activeContent: null, activeExtra: 0 })
|
||||
})
|
||||
|
||||
it('is empty-safe', () => {
|
||||
expect(planSummary([])).toEqual({ done: 0, total: 0, activeContent: null, activeExtra: 0 })
|
||||
})
|
||||
})
|
||||
|
||||
describe('TodoPanel', () => {
|
||||
it('renders nothing while the list is empty', () => {
|
||||
const { container } = render(<TodoPanel todos={[]} t={t} />)
|
||||
@@ -80,6 +127,18 @@ describe('TodoPanel', () => {
|
||||
expect(screen.getAllByRole('listitem')).toHaveLength(3)
|
||||
})
|
||||
|
||||
it('marks every parallel active item, and counts them all in the header', () => {
|
||||
render(<TodoPanel todos={PARALLEL} t={t} />)
|
||||
fireEvent.click(screen.getByRole('button', { expanded: false }))
|
||||
// The old unconditional cap made this list unreachable: three items carry
|
||||
// the in-progress glyph at once, and the header counts all three.
|
||||
const statuses = screen.getAllByRole('listitem').map(li => li.getAttribute('data-status'))
|
||||
expect(statuses.filter(s => s === 'in_progress')).toHaveLength(3)
|
||||
expect(screen.getByText('跑后台构建')).toBeTruthy()
|
||||
expect(screen.getByText('读源码')).toBeTruthy()
|
||||
expect(screen.getByText('1 已完成 · 3 进行中 · 1 待处理')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('an all-completed list collapses the summary to the done count alone', () => {
|
||||
render(<TodoPanel todos={[{ content: '都完了', status: 'completed' }]} t={t} />)
|
||||
expect(screen.getByRole('button', { expanded: false })).toBeTruthy()
|
||||
@@ -145,12 +204,30 @@ describe('TodoRow', () => {
|
||||
expect(screen.getByText('1/3 已完成 · 写组件')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('reports the extra active count outside the ellipsized summary text', () => {
|
||||
const { container } = render(<TodoRow {...rowProps(resultNode(JSON.stringify({ todos: PARALLEL })))} />)
|
||||
const text = screen.getByText('1/5 已完成 · 写组件')
|
||||
const extra = screen.getByText('+2')
|
||||
// Separate spans: .summary truncates, the count must not travel inside it.
|
||||
expect(text.contains(extra)).toBe(false)
|
||||
expect(container.textContent).toContain('1/5 已完成 · 写组件+2')
|
||||
})
|
||||
|
||||
it('omits the active clause when no item is in progress and reads running-call args', () => {
|
||||
const args = JSON.stringify({ todos: [{ content: 'x', status: 'completed' }] })
|
||||
render(<TodoRow {...rowProps({ callId: 'c1', name: 'todo_write', argsRaw: args, turn: 1, step: 1, time: 1_000, callView: null })} />)
|
||||
expect(screen.getByText('1/1 已完成')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('keeps the counts when an active item has unusable content, instead of the generic summary', () => {
|
||||
// planSummary yields activeContent null here, but the counts are known good,
|
||||
// so the row drops only the active clause — `?? model.summary` never runs.
|
||||
const args = JSON.stringify({ todos: [{ content: 'done', status: 'completed' }, { content: 42, status: 'in_progress' }] })
|
||||
const { container } = render(<TodoRow {...rowProps(resultNode(args))} />)
|
||||
expect(screen.getByText('1/2 已完成')).toBeTruthy()
|
||||
expect(container.textContent).not.toContain('+')
|
||||
})
|
||||
|
||||
it('keeps the non-ok execution states visible through the shared row states', () => {
|
||||
// A running call (no result yet) carries the running state (row sweep).
|
||||
const args = JSON.stringify({ todos: LIST })
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
/**
|
||||
* The one-line contract of the ToolRow summary line as CSS text. jsdom has no
|
||||
* layout, so the rendering specs (chat-tool-row.spec.tsx) can pin which spans
|
||||
* exist but not whether a narrow row still fits on one line; these read the
|
||||
* declarations the layout depends on.
|
||||
*/
|
||||
import { readFileSync } from 'node:fs'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
import { describe, expect, it } from 'vitest'
|
||||
|
||||
const css = readFileSync(fileURLToPath(new URL('../src/client/chat/ToolRow.module.css', import.meta.url)), 'utf8')
|
||||
/** Declarations only: the sheet's prose names the properties it explains. */
|
||||
const declarationText = css.replace(/\/\*[\s\S]*?\*\//g, ' ')
|
||||
|
||||
function declarations(selector: string): string[] {
|
||||
// Anchored at a rule boundary: an unanchored match would silently read a
|
||||
// compound rule that merely contains the selector (`.root:hover .summarySuffix`)
|
||||
// if one ever lands above the base rule.
|
||||
const rule = new RegExp(`(?:^|\\})\\s*\\${selector}\\s*\\{([^{}]*)\\}`).exec(declarationText)
|
||||
if (rule === null) throw new Error(`ToolRow.module.css has no \`${selector}\` rule`)
|
||||
return (rule[1] ?? '').split(';').map(part => part.trim()).filter(Boolean)
|
||||
}
|
||||
|
||||
describe('ToolRow.module.css summary line', () => {
|
||||
it('keeps the summary suffix on one line and unshrunk', () => {
|
||||
// `flex: none` stops the box shrinking, not the text wrapping: without
|
||||
// `nowrap`, a row too narrow for title + separator + suffix wraps the `+n`
|
||||
// onto a second line — the exact case the slot exists to survive.
|
||||
expect(declarations('.summarySuffix')).toEqual(expect.arrayContaining([
|
||||
'flex: none',
|
||||
'white-space: nowrap',
|
||||
]))
|
||||
})
|
||||
|
||||
it('leaves the truncation to the summary text alone', () => {
|
||||
// The suffix must never ellipsize: a clipped count reads as a smaller
|
||||
// number rather than as missing information.
|
||||
expect(declarations('.summary')).toEqual(expect.arrayContaining([
|
||||
'overflow: hidden',
|
||||
'text-overflow: ellipsis',
|
||||
'white-space: nowrap',
|
||||
]))
|
||||
expect(declarations('.summarySuffix')).not.toEqual(expect.arrayContaining(['text-overflow: ellipsis']))
|
||||
})
|
||||
})
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/code-runtime/README.md
|
||||
README.md: 2d32a05071efdfa05c336211196bb769ed5a5fc7
|
||||
README.zh.md: 62c9a395ac3cf2b5cd55455ab5a273a1e276f6b7
|
||||
README.md: f20a287419b94b1a9dc1d8da7303fc4d3032cfd3
|
||||
README.zh.md: f5cd4c9949f2bd7a7d6d7cd078144910712a3819
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
English | [中文](README.zh.md)
|
||||
|
||||
The code-execution capability seam (see [capability seams](../../.agents/notes/implemented/architecture/2026-06-13-capability-seams.md)): an abstract runtime interface for executing one model-written program against host-provided async bindings, capturing what it printed and returned. The consumer is the tool registry's [Code Mode](../core/tools/README.md) (`tools: { mode: code }` — the `run_code` tool and the generated TypeScript SDK); design in the [Code Mode Agent Note](../../.agents/notes/implemented/feature/2026-06-15-code-mode.md). **Product** packages.
|
||||
The code-execution capability seam (see [capability seams](../../.agents/notes/implemented/architecture/2026-06-13-capability-seams.md)): an abstract runtime interface for executing one model-written program against host-provided async bindings, capturing what it printed and returned. The consumer is the tool registry's [Code Mode](../core/tools/README.md) (`tools: { mode: code }` — the `run_code` tool and the SDK generated in the loaded runtime's `language`); design in the [Code Mode Agent Note](../../.agents/notes/implemented/feature/2026-06-15-code-mode.md). **Product** packages.
|
||||
|
||||
| Package | Role | ctx key |
|
||||
|---|---|---|
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
[English](README.md) | 中文
|
||||
|
||||
代码执行能力 seam(参见[能力 seam](../../.agents/notes/implemented/architecture/2026-06-13-capability-seams.md)):一个抽象运行时接口,用于对宿主提供的异步绑定执行模型编写的程序,并捕获它打印和返回的内容。消费方是工具注册表的 [Code Mode](../core/tools/README.md)(`tools: { mode: code }`,即 `run_code` 工具和生成的 TypeScript SDK);设计见 [Code Mode Agent Note](../../.agents/notes/implemented/feature/2026-06-15-code-mode.md)。这些全是**产品**包。
|
||||
代码执行能力 seam(参见[能力 seam](../../.agents/notes/implemented/architecture/2026-06-13-capability-seams.md)):一个抽象运行时接口,用于对宿主提供的异步绑定执行模型编写的程序,并捕获它打印和返回的内容。消费方是工具注册表的 [Code Mode](../core/tools/README.md)(`tools: { mode: code }`,即 `run_code` 工具和按所加载运行时 `language` 生成的 SDK);设计见 [Code Mode Agent Note](../../.agents/notes/implemented/feature/2026-06-15-code-mode.md)。这些全是**产品**包。
|
||||
|
||||
| 包 | 职责 | ctx key |
|
||||
|---|---|---|
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/code-runtime/code-runtime/README.md
|
||||
README.md: fafaca85ac63e83e16882ef17f9f8995103866fd
|
||||
README.zh.md: dc771bc16557d4b68529f2fde29eef1c09f299b5
|
||||
README.md: bb1c20d00a260f643f601c42c6e48722437d5aab
|
||||
README.zh.md: 15fbcecf77b2318acf3b09101802cd032ae426d2
|
||||
|
||||
@@ -11,7 +11,7 @@ This package is the interface third of the capability (the bash trio is the temp
|
||||
| Member | Semantics |
|
||||
|---|---|
|
||||
| `run(request)` | Execute one program against the request's bindings. **Resolves with an error FIELD for every program outcome** — parse/transform failure, thrown exception, invalid completion, output overflow, budget expiry, abort, or substrate death (`CodeRunFailure`'s orthogonal `kind` taxonomy); it rejects only for caller misuse of the seam itself (e.g. a run submitted after disposal). The program runs as the body of an async function: top-level `await`/`return` work, and a lossless JSON completion becomes `result.value`. |
|
||||
| `language` | Readonly descriptor: the source language `run` expects (`'typescript'` is the well-known value). Informational, not gating — a consumer that generates language-specific presentation switches on it and fails loud on a language it cannot present. |
|
||||
| `language` | Readonly descriptor: the source language `run` expects. `'typescript'` and `'python'` are the well-known values — those `dsh-tools` presents; only `'typescript'` has a published backend. Informational, not gating — a consumer that generates language-specific presentation switches on it and fails loud on a language it cannot present. |
|
||||
| `isolation` | Readonly descriptor: the execution substrate (`'worker-thread'`, `'process'`, `'container'`). A label for deployments and diagnostics, **not a security claim**. |
|
||||
|
||||
Semantics every implementation must honor (contract details in the class JSDoc): binding calls bridge complete lossless-JSON arguments and resolutions with no seam-level byte cap; the program is treated as a hostile peer (arbitrary binding names are own properties, malformed traffic never crashes the host); no state survives between runs; disposal terminates in-flight runs AND awaits their exit before completing.
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
| 成员 | 语义 |
|
||||
|---|---|
|
||||
| `run(request)` | 针对请求的绑定执行一段程序。**所有程序失败结果都通过 resolve 结果中的 error 字段报告**:包括解析/转换失败、抛出异常、无效完成值、输出溢出、预算到期、中止或执行基底终止(由 `CodeRunFailure` 的正交 `kind` 分类表示);只有调用方误用 seam 本身时才 reject(例如 dispose(资源释放)后仍提交运行)。程序作为异步函数的函数体运行,因此顶层 `await`/`return` 可用,无损 JSON 完成值会成为 `result.value`。 |
|
||||
| `language` | 只读描述符:`run` 期望的源语言(已知值为 `'typescript'`)。仅供参考,不作门禁;生成语言专用呈现的消费方会根据该值选择分支,遇到无法呈现的语言时明确失败。 |
|
||||
| `language` | 只读描述符:`run` 期望的源语言。已知值为 `'typescript'` 与 `'python'`——`dsh-tools` 能呈现的那些;其中只有 `'typescript'` 有已发布的后端。仅供参考,不作门禁;生成语言专用呈现的消费方会根据该值选择分支,遇到无法呈现的语言时明确失败。 |
|
||||
| `isolation` | 只读描述符:执行基底(`'worker-thread'`、`'process'`、`'container'`)。供部署与诊断使用,**不构成安全声明**。 |
|
||||
|
||||
每个实现都必须遵守以下语义(完整契约见类 JSDoc):绑定调用会桥接完整的无损 JSON 参数与 resolve 值,seam 层不设字节上限;程序被视为敌对对等方(任意绑定名称都会成为自有属性,格式错误的通信绝不能使宿主崩溃);不同运行之间不保留任何状态;dispose 会终止进行中的运行,并且在完成前等待其退出。
|
||||
|
||||
@@ -107,7 +107,8 @@ export abstract class CodeRuntime extends Service {
|
||||
* lowercase identifier. Informational, not gating — a consumer that
|
||||
* generates language-specific presentation (typed SDK stubs, usage
|
||||
* instructions) switches on it and fails loud on a language it cannot
|
||||
* present. Well-known value: `'typescript'`.
|
||||
* present. Well-known values: `'typescript'` and `'python'`, those
|
||||
* `dsh-tools` presents; only `'typescript'` has a published backend.
|
||||
*/
|
||||
abstract readonly language: string
|
||||
|
||||
|
||||
@@ -161,7 +161,7 @@ export type TurnEndReason = TurnEndReasonMap[keyof TurnEndReasonMap]
|
||||
export interface TodoItem {
|
||||
/** What this task is — a short imperative line shown in the UI. */
|
||||
content: string
|
||||
/** Lifecycle state. `in_progress` marks the single task being worked now. */
|
||||
/** Lifecycle state. `in_progress` marks a task being worked now; parallel work may mark several. */
|
||||
status: 'pending' | 'in_progress' | 'completed'
|
||||
}
|
||||
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/core/tools/README.md
|
||||
README.md: 80ea3cc93437d48a7ea0ffba0ff4d2ef2407755f
|
||||
README.zh.md: 691d2f2fcccdaa1bcab5343b2fce661d9c99e8ad
|
||||
README.md: 81cc57983d83fd19468017b217d4db9978f4e228
|
||||
README.zh.md: 9f875bd80a03d1d0f78625ee98eeaad9d118f871
|
||||
|
||||
@@ -13,7 +13,7 @@ tools:
|
||||
mode: native # native (default) | code | both
|
||||
```
|
||||
|
||||
`native` contributes visible tools as function definitions. `code` contributes the reserved `run_code` transport and generated `tools:sdk` section; `both` contributes both forms. The reserved transport cannot be registered, shadowed, restricted, or removed. Non-native modes require a TypeScript `ctx.codeRuntime`, and a `systemPrompt.toolOrder` entry for a tool the mode does not contribute rejects prompt assembly. A `system-prompt/assemble` listener may replace the registry's contributions; its returned assembly is authoritative, so that listener owns preserving a usable Code Mode protocol.
|
||||
`native` contributes visible tools as function definitions. `code` contributes the reserved `run_code` transport and generated `tools:sdk` section; `both` contributes both forms. The reserved transport cannot be registered, shadowed, restricted, or removed. Non-native modes require a `ctx.codeRuntime` whose `language` has a registered SDK renderer — TypeScript ships via [`dsh-code-runtime-worker`](../../code-runtime/code-runtime-worker/README.md); a Python renderer is built in and drives any runtime that reports `language: 'python'` (a first-party `dsh-code-runtime-python` backend is delivered separately). A runtime language with no renderer fails prompt assembly loudly, and a `systemPrompt.toolOrder` entry for a tool the mode does not contribute rejects prompt assembly. A `system-prompt/assemble` listener may replace the registry's contributions; its returned assembly is authoritative, so that listener owns preserving a usable Code Mode protocol.
|
||||
|
||||
### Public API
|
||||
|
||||
@@ -114,9 +114,9 @@ Returning `undefined` selects generic fallback. Presenters depend only on their
|
||||
|
||||
### Code Mode
|
||||
|
||||
Under `code` or `both`, the registry exposes the reserved `run_code` transport and a deterministic TypeScript SDK for the current scope; only the program's outer logs and return value re-enter model context. The SDK declares exact `ToolArgsMap` and `ToolOutputMap` entries for every visible tool, and each binding resolves to the tool's canonical JSON value. Each lossless-JSON binding call re-enters the complete tool pipeline under the native scheduling contract (concurrency-safe calls may overlap up to `maxParallelSubCalls`; exclusive calls run alone as ordering barriers) with logged correlation to the outer call. Denials and other failed results reject with the real program-visible `ToolCallError` carrying only `toolName` and `message`; Native content and internal error codes stay outside the Code contract. Ordinary side effects are not rolled back, and sub-call `additionalContexts` are deferred through the parent result to preserve call/result adjacency. Run settlement aborts and drains outstanding bindings; runtime failures surface as `CodeRunFailedError`. See the [Code Mode foundation](../../../.agents/notes/implemented/feature/2026-06-15-code-mode.md), [typed-return contract](../../../.agents/notes/implemented/feature/2026-07-20-code-mode-typed-tool-returns.md), and [code-runtime seam](../../code-runtime/README.md). Try `pnpm run demo:code-mode`.
|
||||
Under `code` or `both`, the registry exposes the reserved `run_code` transport and a deterministic SDK for the current scope, generated in the loaded runtime's language — the registry selects the renderer by `ctx.codeRuntime.language` (`typescript` → the TypeScript SDK below, `python` → the Python SDK). Only the program's outer logs and return value re-enter model context. The SDK declares exact per-tool argument and canonical-output types for every visible tool (`ToolArgsMap`/`ToolOutputMap` in TypeScript, named `TypedDict`s in Python), and each binding resolves to the tool's canonical JSON value. Each lossless-JSON binding call re-enters the complete tool pipeline under the native scheduling contract (concurrency-safe calls may overlap up to `maxParallelSubCalls`; exclusive calls run alone as ordering barriers) with logged correlation to the outer call. Denials and other failed results reject with the real program-visible `ToolCallError` carrying only `toolName` and `message`; Native content and internal error codes stay outside the Code contract. Ordinary side effects are not rolled back, and sub-call `additionalContexts` are deferred through the parent result to preserve call/result adjacency. Run settlement aborts and drains outstanding bindings; runtime failures surface as `CodeRunFailedError`. See the [Code Mode foundation](../../../.agents/notes/implemented/feature/2026-06-15-code-mode.md), [typed-return contract](../../../.agents/notes/implemented/feature/2026-07-20-code-mode-typed-tool-returns.md), and [code-runtime seam](../../code-runtime/README.md). Try `pnpm run demo:code-mode`.
|
||||
|
||||
- **The SDK section** (`tools:sdk`, order 150): a lazy prompt section regenerating, at each assembly, `JsonValue`, exact `ToolArgsMap` / `ToolOutputMap`, `ToolName`, the `ToolCallError` declaration, and a mapped `tools` namespace for the calling scope's visible end capabilities (exotic names via quoted keys), plus fixed usage instructions. Deterministic — lexicographic tool order, byte-identical text for an unchanged tool set (prefix-cache-friendly). The codegen (`jsonSchemaToTs`, exported) handles every unified schema construct and degrades unsupported raw constructs to `unknown`, never throwing during prompt assembly.
|
||||
- **The SDK section** (`tools:sdk`, order 150): a lazy prompt section regenerating the language-appropriate SDK text at each assembly. In the TypeScript flavor it emits `JsonValue`, exact `ToolArgsMap` / `ToolOutputMap`, `ToolName`, the `ToolCallError` declaration, and a mapped `tools` namespace for the calling scope's visible end capabilities (exotic names via quoted keys), plus fixed usage instructions; the Python flavor (`ctx.codeRuntime.language === 'python'`) emits the equivalent named `TypedDict`s and a `tools` object with matching usage instructions. Deterministic — lexicographic tool order, byte-identical text for an unchanged tool set (prefix-cache-friendly). Both codegens are exported and never throw during prompt assembly: `jsonSchemaToTs` handles every unified schema construct and degrades unsupported raw constructs to `unknown`; `jsonSchemaToPy` does the same, degrading to `Any` (and a whole object to `dict[str, Any]` when a field name is not a legal `TypedDict` attribute, or whenever it is called outside the SDK render, which supplies the naming context a `TypedDict` declaration needs).
|
||||
- **The dispatch bridge** (`run_code`'s execute): every binding call is snapshotted as lossless JSON before dispatch (`undefined`, `BigInt`, cycles, sparse arrays, `-0`, and exotic objects reject that one call), scheduled through a per-run pool that reuses the native concurrency contract — calls start strictly in submission order, consecutive `isConcurrencySafe` calls overlap up to the validated `maxParallelSubCalls` config (default 10; `1` restores serial dispatch), and an exclusive-classified call drains the pool, runs alone, and bars later calls — given the outer execution's opaque token as `parent`, and run through the complete pre-execute → guards → execute → post-execute → result pipeline. A success returns the final canonical value after policy; a failure reaches the worker as one message and becomes `ToolCallError(toolName, message)`. Each started sub-call logs a `tool/code-dispatch-start` event (deterministic id `<parent>:code:<n>`, numbered by submission) at pipeline entry and settles with one `tool/code-dispatch` event carrying the complete model-facing `content`/`isError` outcome (the `tool/result` vocabulary, so UIs render sub-calls through the native path — the pair's `time` fields carry per-sub-call timing); a queued call abandoned by run settlement logs neither. `deriveMessages()` surfaces neither event nor persists the canonical value. Token correlation lets commit-style observers defer an inner success until the final `run_code` result without exposing the live outer execution; ordinary tool side effects are not rolled back. Every sub-call `additionalContexts` entry is deferred through the outer `ToolRunContext` in dispatch order; the loop appends those contexts only after the parent `run_code` result, preserving adjacency and retaining each source/meta even when the program later fails.
|
||||
- **Settlement discipline**: the bridge owns a run-scoped abort that follows the outer signal in and fires when the run settles for any reason, so a budget expiry aborts an in-flight sub-tool instead of orphaning it; the bridge then drains its queue BEFORE returning, so every `tool/code-dispatch` lands inside the open turn. A failed run throws `CodeRunFailedError` (`code: 'CODE_RUN_FAILED'`, message = the failure kind + captured logs), which the pipeline converts to a structured `isError` the model self-corrects from.
|
||||
- **Result boundary**: intermediate binding values cross the worker boundary whole and have no per-binding byte cap. `run_code` returns canonical `{ logs: string[], result?: JsonValue }`; strings render raw, every other present JSON root renders through a stack-safe pretty JSON traversal whose total indentation is capped at ten characters (deeper subtrees stay compact), `null` remains explicit, and absent `result` means the program returned `undefined`. The worker's configurable `maxOutputBytes` (default 64 MiB) applies only to the combined serialized outer log-array, completion-value, or failure-message payloads; fixed result-envelope syntax and presentation whitespace are outside that ledger. Invalid and over-limit completions fail explicitly, and only this outer result is eligible for ordinary spill.
|
||||
@@ -145,7 +145,7 @@ Prefix-stable while visible definitions and their order are unchanged. Registrat
|
||||
|
||||
#### What the model sees
|
||||
|
||||
Code Mode exposes the generated [`run_code` schema](../../../docs/tool-catalog.md#deepseek-aidsh-tools), the SDK instructions below, and the generated exact `declare const tools` block. `both` exposes normal schemas and this Code Mode surface.
|
||||
Code Mode exposes the generated [`run_code` schema](../../../docs/tool-catalog.md#deepseek-aidsh-tools), the SDK instructions below, and the generated exact SDK block for the loaded runtime's language (the TypeScript `declare const tools` block, or the Python `tools` declaration). `both` exposes normal schemas and this Code Mode surface. The instructions and SDK block match the loaded runtime's language; the TypeScript flavor (via [`dsh-code-runtime-worker`](../../code-runtime/code-runtime-worker/README.md)) is shown below, and the Python flavor (for any runtime reporting `language: 'python'`) is the same shape with Python syntax (`await tools.name(args)`, subscript access for exotic names, `print(...)` and top-level `return`).
|
||||
|
||||
##### Code Mode SDK instructions
|
||||
|
||||
@@ -190,6 +190,6 @@ Append-only; newly visible content follows the reusable request prefix and does
|
||||
- **`tools/pre-execute` deliberately cannot rewrite `exec.arguments`** — logged and rendered args would desync from what ran; the rewrite design is [a proposed Agent Note](../../../.agents/notes/proposed/feature/2026-06-30-pre-tool-input-rewrite.md).
|
||||
- **Caller-defined subagent and workflow structured outputs remain object-rooted** — this is a consumer-level guard; the shared schema vocabulary and tool outputs support every JSON root.
|
||||
- **`timeoutMs` on a definition is declarative only** — the registry never enforces deadlines; enforcement requires the `@deepseek-ai/dsh-timeout-policy` wrapper.
|
||||
- **Code Mode is TypeScript-only and the presentation mode is service-wide** — `mode: code`/`both` rejects prompt assembly unless `ctx.codeRuntime.language === 'typescript'`; scoped restrictions/shadows still choose each agent's visible bindings, but one tool cannot be native-only while another is code-only.
|
||||
- **Code Mode's SDK language follows the one loaded runtime and the presentation mode is service-wide** — `mode: code`/`both` rejects prompt assembly unless `ctx.codeRuntime.language` has a registered SDK renderer (`typescript` via the worker backend, `python` for any runtime reporting that language); scoped restrictions/shadows still choose each agent's visible bindings, but one tool cannot be native-only while another is code-only, and a single runtime fixes the language service-wide (the [language-dispatch Agent Note](../../../.agents/notes/implemented/feature/2026-07-31-code-mode-language-dispatch.md) owns the lookup, and why the registry reads the loaded runtime instead of carrying a language field of its own).
|
||||
- **Code Mode intermediate values are execution-local and unbounded by bytes** — the canonical typed values cannot be reconstructed from session replay and may exhaust process or worker memory; only the outer `run_code` output has the worker's configurable hard cap. The durable log copy of each sub-call IS bounded: the `tools/code-dispatch-log` waterfall lets the spill policy replace an oversized `tool/code-dispatch` content with a preview + locator ([rationale](../../../.agents/notes/implemented/feature/2026-07-26-code-dispatch-log-spill.md)).
|
||||
- **`run_code` state is fresh per run** — a persistent REPL-style kernel is rejected for the MVP (cross-call state would be invisible to the log); see [the Code Mode Agent Note](../../../.agents/notes/implemented/feature/2026-06-15-code-mode.md).
|
||||
|
||||
@@ -13,7 +13,7 @@ tools:
|
||||
mode: native # native (default) | code | both
|
||||
```
|
||||
|
||||
`native` 以函数定义的形式贡献可见工具。`code` 贡献保留的 `run_code` 传输和生成的 `tools:sdk` 段;`both` 同时贡献两种形式。不能注册、遮蔽、限制或移除该保留传输。非原生模式要求存在 TypeScript `ctx.codeRuntime`;如果 `systemPrompt.toolOrder` 条目指向当前模式未贡献的工具,系统会拒绝组装提示词。`system-prompt/assemble` 监听器可以替换注册表贡献;它返回的组装结果具有权威性,因此该监听器负责保留可用的 Code Mode 协议。
|
||||
`native` 以函数定义的形式贡献可见工具。`code` 贡献保留的 `run_code` 传输和生成的 `tools:sdk` 段;`both` 同时贡献两种形式。不能注册、遮蔽、限制或移除该保留传输。非原生模式要求所加载 `ctx.codeRuntime` 的 `language` 有已注册的 SDK 渲染器——TypeScript 经 [`dsh-code-runtime-worker`](../../code-runtime/code-runtime-worker/README.md) 交付;Python 渲染器内置,驱动任何报告 `language: 'python'` 的运行时(第一方 `dsh-code-runtime-python` 后端另行交付)。没有渲染器的运行时语言会让提示词组装响亮失败;如果 `systemPrompt.toolOrder` 条目指向当前模式未贡献的工具,系统会拒绝组装提示词。`system-prompt/assemble` 监听器可以替换注册表贡献;它返回的组装结果具有权威性,因此该监听器负责保留可用的 Code Mode 协议。
|
||||
|
||||
### 公开 API
|
||||
|
||||
@@ -114,9 +114,9 @@ ctx.tools.register(defineTool({
|
||||
|
||||
### Code Mode
|
||||
|
||||
在 `code` 或 `both` 模式下,注册表为当前作用域公开保留的 `run_code` 传输和确定性的 TypeScript SDK;只有程序的外层日志与返回值会重新进入模型上下文。SDK 为每个可见工具声明精确的 `ToolArgsMap` 和 `ToolOutputMap` 条目,每个绑定都会解析为该工具的规范 JSON 值。每个无损 JSON 绑定调用都会在原生调度契约下重新进入完整工具流水线(并发安全的调用最多可重叠 `maxParallelSubCalls` 个;独占调用单独运行并构成排序屏障),并在日志中与外层调用建立关联。拒绝及其他失败结果会以程序实际可见的 `ToolCallError` 形式拒绝,且只携带 `toolName` 和 `message`;Native 内容和内部错误码留在 Code 契约之外。普通副作用不会回滚,子调用的 `additionalContexts` 会通过父结果延迟,以保持调用/结果相邻。运行结算会中止并排空尚未完成的绑定;运行时失败以 `CodeRunFailedError` 形式出现。参见 [Code Mode 基础](../../../.agents/notes/implemented/feature/2026-06-15-code-mode.md)、[类型化返回契约](../../../.agents/notes/implemented/feature/2026-07-20-code-mode-typed-tool-returns.md)和[代码运行时 seam](../../code-runtime/README.md)。可以运行 `pnpm run demo:code-mode` 试用。
|
||||
在 `code` 或 `both` 模式下,注册表为当前作用域公开保留的 `run_code` 传输和按所加载运行时语言生成的确定性 SDK——注册表按 `ctx.codeRuntime.language` 选择渲染器(`typescript` → 下方的 TypeScript SDK,`python` → Python SDK)。只有程序的外层日志与返回值会重新进入模型上下文。SDK 为每个可见工具声明精确的参数与规范输出类型(TypeScript 为 `ToolArgsMap`/`ToolOutputMap`,Python 为具名 `TypedDict`),每个绑定都会解析为该工具的规范 JSON 值。每个无损 JSON 绑定调用都会在原生调度契约下重新进入完整工具流水线(并发安全的调用最多可重叠 `maxParallelSubCalls` 个;独占调用单独运行并构成排序屏障),并在日志中与外层调用建立关联。拒绝及其他失败结果会以程序实际可见的 `ToolCallError` 形式拒绝,且只携带 `toolName` 和 `message`;Native 内容和内部错误码留在 Code 契约之外。普通副作用不会回滚,子调用的 `additionalContexts` 会通过父结果延迟,以保持调用/结果相邻。运行结算会中止并排空尚未完成的绑定;运行时失败以 `CodeRunFailedError` 形式出现。参见 [Code Mode 基础](../../../.agents/notes/implemented/feature/2026-06-15-code-mode.md)、[类型化返回契约](../../../.agents/notes/implemented/feature/2026-07-20-code-mode-typed-tool-returns.md)和[代码运行时 seam](../../code-runtime/README.md)。可以运行 `pnpm run demo:code-mode` 试用。
|
||||
|
||||
- **SDK 段**(`tools:sdk`,顺序 150):一个惰性提示词段,每次组装时都会重新生成 `JsonValue`、精确的 `ToolArgsMap` / `ToolOutputMap`、`ToolName`、`ToolCallError` 声明、面向调用作用域可见最终能力的映射 `tools` 命名空间(特殊名称使用带引号的键),以及固定用法说明。其输出具有确定性:工具按字典序排列;工具集合不变时,文本逐字节相同(有利于前缀 cache)。导出的代码生成器 `jsonSchemaToTs` 会处理统一 schema 的每种构造,并将不受支持的原始构造降级为 `unknown`,绝不会在提示词组装期间抛出。
|
||||
- **SDK 段**(`tools:sdk`,顺序 150):一个惰性提示词段,每次组装时都会重新生成与所加载运行时语言相符的 SDK 文本。TypeScript 形态发出 `JsonValue`、精确的 `ToolArgsMap` / `ToolOutputMap`、`ToolName`、`ToolCallError` 声明、面向调用作用域可见最终能力的映射 `tools` 命名空间(特殊名称使用带引号的键),以及固定用法说明;Python 形态(`ctx.codeRuntime.language === 'python'`)发出等价的具名 `TypedDict` 与一个带相同用法说明的 `tools` 对象。其输出具有确定性:工具按字典序排列;工具集合不变时,文本逐字节相同(有利于前缀 cache)。两个代码生成器都已导出,且绝不会在提示词组装期间抛出:`jsonSchemaToTs` 处理统一 schema 的每种构造并将不受支持的原始构造降级为 `unknown`;`jsonSchemaToPy` 同理,降级为 `Any`(当某字段名不是合法的 `TypedDict` 属性时,或在 SDK 渲染之外被调用时——`TypedDict` 声明所需的命名上下文由该渲染提供——整个对象降级为 `dict[str, Any]`)。
|
||||
- **分发桥接层**(`run_code` 的 execute):每个绑定调用都会在分发前快照为无损 JSON(`undefined`、`BigInt`、循环、稀疏数组、`-0` 和特殊对象会使该次调用被拒绝),经由每次运行独有、复用原生并发契约的池调度——调用严格按提交顺序启动,连续的 `isConcurrencySafe` 调用最多可重叠经校验的 `maxParallelSubCalls` 配置个(默认 10;设为 `1` 即恢复串行分发),被分类为独占的调用先排空池、单独运行并阻挡其后的调用——以外层执行的不透明 token 作为 `parent`,并经过完整的 pre-execute → guards → execute → post-execute → result 流水线。成功会返回策略处理后的最终规范值;失败以一条消息到达 worker,并成为 `ToolCallError(toolName, message)`。每个已启动的子调用在进入流水线时记录一条 `tool/code-dispatch-start` 事件(确定性 id `<parent>:code:<n>`,按提交顺序编号),并以一条携带完整模型可见 `content`/`isError` 结果的 `tool/code-dispatch` 事件完结(采用 `tool/result` 词汇,因此 UI 会沿原生路径呈现子调用——这对事件的 `time` 字段承载每个子调用的计时);因 run 结算而被放弃的排队调用两者都不记录。`deriveMessages()` 既不公开这两个事件,也不持久化规范值。token 关联让以提交为语义的观察器能够把内部成功延迟到最终 `run_code` 结果,而无需公开实时外层执行;普通工具副作用不会回滚。每个子调用的 `additionalContexts` 条目都会按分发顺序通过外层 `ToolRunContext` 延迟;循环只在父级 `run_code` 结果之后追加这些上下文,从而保持相邻关系,并且即使程序后来失败,也会保留各自的来源/元数据。
|
||||
- **结算纪律**:桥接层拥有一个运行作用域的中止机制;该中止会跟随传入的外层信号,并在运行因任何原因结算时触发,因此预算耗尽会中止正在运行的子工具,而不会将其遗留。桥接层随后会在返回之前排空队列,使每个 `tool/code-dispatch` 都落在仍打开的轮次内。失败的运行会抛出 `CodeRunFailedError`(`code: 'CODE_RUN_FAILED'`,message = 失败类型 + 已捕获日志),流水线会将其转换为模型可据以自我修正的结构化 `isError`。
|
||||
- **结果边界**:中间绑定值会完整跨越 worker 边界,且没有逐绑定字节上限。`run_code` 返回规范的 `{ logs: string[], result?: JsonValue }`;字符串原样呈现,其他所有存在的 JSON 根都通过栈安全的美化 JSON 遍历呈现,总缩进最多为 10 个字符(更深的子树保持紧凑),`null` 保持显式,而缺少 `result` 表示程序返回 `undefined`。worker 可配置的 `maxOutputBytes`(默认 64 MiB)只应用于组合序列化后的外层日志数组、完成值或失败消息载荷;固定的结果 envelope 语法和呈现空白不计入该账本。无效和超限的完成会明确失败,只有此外层结果可以使用普通 spill。
|
||||
@@ -145,7 +145,7 @@ agent loop 将连续的 `parallel` 调用归入有界滚动池,并把每个 `e
|
||||
|
||||
#### 模型看到的内容
|
||||
|
||||
Code Mode 会公开生成的 [`run_code` schema](../../../docs/tool-catalog.md#deepseek-aidsh-tools)、下方 SDK 说明,以及生成的精确 `declare const tools` 块。`both` 会同时公开普通 schema 与此 Code Mode 接口。
|
||||
Code Mode 会公开生成的 [`run_code` schema](../../../docs/tool-catalog.md#deepseek-aidsh-tools)、下方 SDK 说明,以及按所加载运行时语言生成的精确 SDK 块(TypeScript 的 `declare const tools` 块,或 Python 的 `tools` 声明)。`both` 会同时公开普通 schema 与此 Code Mode 接口。说明与 SDK 块随所加载运行时的语言切换;下方展示 TypeScript 风格(经 [`dsh-code-runtime-worker`](../../code-runtime/code-runtime-worker/README.md)),Python 风格(用于任何报告 `language: 'python'` 的运行时)形状相同,只是换成 Python 语法(`await tools.name(args)`、特殊名称用下标访问、`print(...)` 与顶层 `return`)。
|
||||
|
||||
##### Code Mode SDK 说明
|
||||
|
||||
@@ -190,6 +190,6 @@ The available tools:
|
||||
- **`tools/pre-execute` 有意不允许改写 `exec.arguments`**:否则日志记录和呈现的参数会与实际运行内容失去同步;改写设计记录在[拟议的 Agent Note](../../../.agents/notes/proposed/feature/2026-06-30-pre-tool-input-rewrite.md)中。
|
||||
- **调用方定义的 subagent 与工作流结构化输出仍要求对象根**:这是消费方层面的守卫;共享 schema 词汇和工具输出支持任意 JSON 根。
|
||||
- **定义上的 `timeoutMs` 仅为声明**:注册表绝不会强制执行截止时间;要强制执行,必须使用 `@deepseek-ai/dsh-timeout-policy` 包装层。
|
||||
- **Code Mode 只支持 TypeScript,且呈现模式在服务内统一**:`mode: code`/`both` 会拒绝组装提示词,除非 `ctx.codeRuntime.language === 'typescript'`;作用域限制/遮蔽仍会选择每个 agent 的可见绑定,但不能让一个工具仅使用 Native,而另一个仅使用 Code。
|
||||
- **Code Mode 的 SDK 语言跟随唯一加载的运行时,且呈现模式在服务内统一**:`mode: code`/`both` 会拒绝组装提示词,除非 `ctx.codeRuntime.language` 有已注册的 SDK 渲染器(`typescript` 经 worker 后端,`python` 用于任何报告该语言的运行时);作用域限制/遮蔽仍会选择每个 agent 的可见绑定,但不能让一个工具仅使用 Native、另一个仅使用 Code,且单个运行时把语言固定为服务级([语言分发 Agent Note](../../../.agents/notes/implemented/feature/2026-07-31-code-mode-language-dispatch.md) 负责这次查表,以及注册表为何读取所加载的运行时而不自带 language 字段)。
|
||||
- **Code Mode 中间值只存在于执行局部,且没有字节上限**:这些规范的类型化值无法从会话回放重建,并可能耗尽进程或 worker 内存;只有外层 `run_code` 输出受 worker 可配置的硬上限约束。每个子调用的持久日志副本则确实有上限:`tools/code-dispatch-log` waterfall 允许 spill 策略把过大的 `tool/code-dispatch` 内容替换为预览加定位符([原理](../../../.agents/notes/implemented/feature/2026-07-26-code-dispatch-log-spill.md))。
|
||||
- **每次运行都会获得全新的 `run_code` 状态**:MVP 不采用持久 REPL 风格内核(跨调用状态不会出现在日志中);参见 [Code Mode Agent Note](../../../.agents/notes/implemented/feature/2026-06-15-code-mode.md)。
|
||||
|
||||
@@ -11,7 +11,7 @@ import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
||||
import type { CodeBindingFunction, CodeRunResult, CodeRuntime } from '@deepseek-ai/dsh-code-runtime'
|
||||
import { snapshotJsonValue } from '@deepseek-ai/dsh-session'
|
||||
import type { JsonValue } from '@deepseek-ai/dsh-session'
|
||||
import { defineTool } from './schema.ts'
|
||||
import { defineTool, parameterSchemaSpecToJsonSchema } from './schema.ts'
|
||||
import { TOOL_REGISTRY_SCHEDULER } from './index.ts'
|
||||
import type { CodeDispatchLog, ToolDefinition, ToolExecutionResult, ToolRegistry, ToolRunContext } from './index.ts'
|
||||
|
||||
@@ -56,6 +56,111 @@ export const RUN_CODE_NAME = 'run_code'
|
||||
/** The `tools:sdk` section order: inside the 100–199 tool-guidance band, after per-tool guidance sections. */
|
||||
export const SDK_SECTION_ORDER = 150
|
||||
|
||||
/**
|
||||
* The language-specific `run_code` schema text: the tool `description` and its
|
||||
* `code` parameter description, kept together so a language's two model-facing
|
||||
* strings share one source of truth. Keyed by `CodeRuntime.language`, mirroring
|
||||
* `SDK_RENDERERS` in {@link ./index.ts}. The emitted flavor MUST match the
|
||||
* semantics the same language's SDK instructions promise, so the model never
|
||||
* receives a TypeScript-shaped schema beside a Python SDK (or vice versa).
|
||||
*/
|
||||
interface RunCodeFlavor {
|
||||
/** The tool `description` the model sees for this language. */
|
||||
readonly description: string
|
||||
/** The `code` parameter's description for this language. */
|
||||
readonly codeDescription: string
|
||||
}
|
||||
|
||||
/**
|
||||
* The TypeScript flavor: the historical default, and the fallback for a schema
|
||||
* read with no runtime mounted ({@link resolveFlavor} owns which readers reach
|
||||
* that). A real assembly always resolves a runtime first, so the model never
|
||||
* sees this fallback outside its own language.
|
||||
*/
|
||||
const TYPESCRIPT_FLAVOR: RunCodeFlavor = {
|
||||
description:
|
||||
'Execute a TypeScript program against the available tools. Write the BODY of an '
|
||||
+ 'async function (erasable syntax only; top-level `await` and `return` work) and '
|
||||
+ 'call tools as `await tools.name(args)` per the declarations in the system prompt. '
|
||||
+ 'Only what you print or return comes back — curate it.',
|
||||
codeDescription: 'The program: the body of an async TypeScript function.',
|
||||
}
|
||||
|
||||
/**
|
||||
* The Python flavor: the body of an async function, top-level `await` and
|
||||
* `return`, answer via `print` and/or the returned value, matching
|
||||
* {@link ./py-types.ts}'s SDK instructions.
|
||||
*/
|
||||
const PYTHON_FLAVOR: RunCodeFlavor = {
|
||||
description:
|
||||
'Execute a Python program against the available tools. Write the BODY of an '
|
||||
+ 'async function (top-level `await` and `return` work) and call tools as '
|
||||
+ '`await tools.name(args)` per the declarations in the system prompt. Answer '
|
||||
+ 'with `print(...)` and/or `return <value>` — only that comes back, so curate it.',
|
||||
codeDescription: 'The program: the body of an async Python function.',
|
||||
}
|
||||
|
||||
/**
|
||||
* The languages Code Mode ships a presentation for. Both per-language tables —
|
||||
* {@link RUN_CODE_FLAVORS} here and `SDK_RENDERERS` in {@link ./index.ts} — are
|
||||
* checked against this union with `satisfies`, so a language added to one and
|
||||
* not the other fails `typecheck` instead of waiting for a runtime that reports
|
||||
* it. The tables stay declared `Record<string, …>` because `CodeRuntime.language`
|
||||
* is an unconstrained `string`: this union pins what the harness ships, while the
|
||||
* `Object.hasOwn` guards reject what a mounted runtime may report.
|
||||
*/
|
||||
export type CodeSdkLanguage = 'typescript' | 'python'
|
||||
|
||||
/** Per-language `run_code` schema flavors (see {@link RunCodeFlavor}); one entry per {@link CodeSdkLanguage}. */
|
||||
const RUN_CODE_FLAVORS: Record<string, RunCodeFlavor> = {
|
||||
typescript: TYPESCRIPT_FLAVOR,
|
||||
python: PYTHON_FLAVOR,
|
||||
} satisfies Record<CodeSdkLanguage, RunCodeFlavor>
|
||||
|
||||
/**
|
||||
* The `description` parameter's model-facing description: language-independent
|
||||
* (the UI label contract is the same for every runtime), shared between the
|
||||
* static spec and the language-aware `parameters` getter so the two emissions
|
||||
* can never drift.
|
||||
*/
|
||||
const RUN_CODE_DESCRIPTION_PARAM_DESCRIPTION
|
||||
= 'Clear, concise description of what this program does in active voice, '
|
||||
+ '5-10 words (shown in the UI). Examples: "Count TODO markers across packages"; '
|
||||
+ '"Read failing test and its fixture"; "Rename config key in every cordis.yml".'
|
||||
|
||||
/**
|
||||
* Resolve the {@link RunCodeFlavor} for the loaded runtime's language, read at
|
||||
* schema-emission time so the model-visible `run_code` schema always matches
|
||||
* the SDK section's language. `peekRuntime` returns `undefined` only when no
|
||||
* runtime is mounted, which reaches this function through definition readers
|
||||
* and `schemas()` — the doc-catalog harvest is the only shipped one, and none
|
||||
* of them feeds a model, because `wireSchemas` calls `requireCodeRuntime`
|
||||
* before projecting — so that path degrades to {@link TYPESCRIPT_FLAVOR}. A
|
||||
* mounted runtime whose language has no flavor entry fails loud, exactly as
|
||||
* `requireCodeRuntime` rejects it at assembly. Keeping this table in step with
|
||||
* `SDK_RENDERERS` is the compiler's job ({@link CodeSdkLanguage}); what this
|
||||
* guard owns is the runtime-supplied language neither table knows, which never
|
||||
* yields a wrong-language schema for a real runtime.
|
||||
*/
|
||||
function resolveFlavor(peekRuntime: () => CodeRuntime | undefined): RunCodeFlavor {
|
||||
const runtime = peekRuntime()
|
||||
if (runtime === undefined) {
|
||||
// No runtime mounted: reached by definition readers and `schemas()`, of
|
||||
// which the doc-catalog harvest is the only shipped one. None feeds a
|
||||
// model — `wireSchemas` calls `requireCodeRuntime` before projecting, so
|
||||
// the assembly path never arrives here. Degrade to the TS default.
|
||||
return TYPESCRIPT_FLAVOR
|
||||
}
|
||||
// Own-property read: a language like `toString`/`constructor` would otherwise
|
||||
// resolve an inherited Object.prototype member as a flavor.
|
||||
const flavor = RUN_CODE_FLAVORS[runtime.language]
|
||||
if (!Object.hasOwn(RUN_CODE_FLAVORS, runtime.language) || flavor === undefined) {
|
||||
const known = Object.keys(RUN_CODE_FLAVORS).map(name => JSON.stringify(name)).join(', ')
|
||||
throw new Error(`dsh-tools: no run_code schema flavor registered for runtime language ${JSON.stringify(runtime.language)} (known: ${known})`)
|
||||
}
|
||||
return flavor
|
||||
}
|
||||
|
||||
/**
|
||||
* Thrown by `run_code` when the program run itself failed — a program
|
||||
* exception, a budget expiry, an abort, or substrate death. Extends
|
||||
@@ -194,6 +299,13 @@ type RunCodeOutput = { logs: string[]; result?: JsonValue }
|
||||
export interface RunCodeBridgeOptions {
|
||||
/** Resolves `ctx.codeRuntime` or throws the loud misconfiguration error (shared with the registry's assembly-time checks). */
|
||||
requireRuntime: () => CodeRuntime
|
||||
/**
|
||||
* Reads `ctx.codeRuntime` without throwing: `undefined` when none is mounted.
|
||||
* Lets schema emission tell "no runtime" (degrade to TS; the readers that
|
||||
* reach it are {@link resolveFlavor}'s) apart from "unknown language" (fail
|
||||
* loud).
|
||||
*/
|
||||
peekRuntime: () => CodeRuntime | undefined
|
||||
/** The run's overlap cap for parallel-classified sub-calls (the registry passes its validated `maxParallelSubCalls`). */
|
||||
maxParallel: number
|
||||
/** Runs the contained `tools/code-dispatch-log` waterfall over one settled sub-dispatch (the registry's private invoker). */
|
||||
@@ -212,22 +324,22 @@ export interface RunCodeBridgeOptions {
|
||||
* @returns the registry-ready definition.
|
||||
*/
|
||||
export function createRunCodeTool(registry: ToolRegistry, options: RunCodeBridgeOptions): ToolDefinition {
|
||||
const { requireRuntime, maxParallel, shapeDispatchLog } = options
|
||||
return defineTool({
|
||||
const { requireRuntime, peekRuntime, maxParallel, shapeDispatchLog } = options
|
||||
const definition = defineTool({
|
||||
name: RUN_CODE_NAME,
|
||||
description:
|
||||
'Execute a TypeScript program against the available tools. Write the BODY of an '
|
||||
+ 'async function (erasable syntax only; top-level `await` and `return` work) and '
|
||||
+ 'call tools as `await tools.name(args)` per the declarations in the system prompt. '
|
||||
+ 'Only what you print or return comes back — curate it.',
|
||||
// The description and `code` parameter description are placeholders here:
|
||||
// the language-aware getters installed below replace both, resolving the
|
||||
// loaded runtime's flavor at schema-emission time so the schema the MODEL
|
||||
// sees matches the SDK section's language. Argument VALIDATION still keys
|
||||
// off this static spec (defineTool closes over it), which is language-
|
||||
// independent (one required string `code`).
|
||||
description: TYPESCRIPT_FLAVOR.description,
|
||||
parameters: {
|
||||
code: { type: 'string', required: true, description: 'The program: the body of an async TypeScript function.' },
|
||||
code: { type: 'string', required: true, description: TYPESCRIPT_FLAVOR.codeDescription },
|
||||
description: {
|
||||
type: 'string',
|
||||
required: true,
|
||||
description: 'Clear, concise description of what this program does in active voice, '
|
||||
+ '5-10 words (shown in the UI). Examples: "Count TODO markers across packages"; '
|
||||
+ '"Read failing test and its fixture"; "Rename config key in every cordis.yml".',
|
||||
description: RUN_CODE_DESCRIPTION_PARAM_DESCRIPTION,
|
||||
},
|
||||
},
|
||||
output: {
|
||||
@@ -569,4 +681,22 @@ export function createRunCodeTool(registry: ToolRegistry, options: RunCodeBridge
|
||||
// title and reads durable result content without duplicating a large raw
|
||||
// result into the host view payload.
|
||||
})
|
||||
// Resolve the language flavor lazily, at the moment the registry projects the
|
||||
// schema (`schemaOf` destructures `description`/`parameters`). The definition
|
||||
// is minted once at registration, before a runtime is known; deferring here
|
||||
// is the least invasive point that still emits the loaded runtime's language.
|
||||
Object.defineProperty(definition, 'description', {
|
||||
enumerable: true,
|
||||
get: () => resolveFlavor(peekRuntime).description,
|
||||
})
|
||||
Object.defineProperty(definition, 'parameters', {
|
||||
enumerable: true,
|
||||
// Recompile through the same spec→schema projection defineTool used, so
|
||||
// the emitted shape can never drift from the validated one.
|
||||
get: () => parameterSchemaSpecToJsonSchema({
|
||||
code: { type: 'string', required: true, description: resolveFlavor(peekRuntime).codeDescription },
|
||||
description: { type: 'string', required: true, description: RUN_CODE_DESCRIPTION_PARAM_DESCRIPTION },
|
||||
}) as unknown as Record<string, unknown>,
|
||||
})
|
||||
return definition
|
||||
}
|
||||
|
||||
@@ -22,8 +22,31 @@ import type { ToolCallView, ToolResultView } from './presentation.ts'
|
||||
import { assertSupportedJsonSchema, validateJsonSchemaValue } from './json-schema.ts'
|
||||
import type { JsonSchemaNode } from './json-schema.ts'
|
||||
import { createRunCodeTool, RUN_CODE_NAME, SDK_SECTION_ORDER } from './code-mode.ts'
|
||||
import type { CodeSdkLanguage } from './code-mode.ts'
|
||||
import { renderToolsSdk } from './ts-types.ts'
|
||||
import type { ToolSdkSchema } from './ts-types.ts'
|
||||
import { renderToolsSdkPy } from './py-types.ts'
|
||||
|
||||
/**
|
||||
* Language → SDK-section renderer. The registry looks up the loaded
|
||||
* `ctx.codeRuntime.language` in this table when assembling the `tools:sdk`
|
||||
* section under a non-native mode; a runtime whose language is not a key
|
||||
* fails the assembly loudly (same idiom as `toolOrder` violations). Adding a
|
||||
* new backend language is three parallel edits — a {@link CodeSdkLanguage}
|
||||
* member, an entry here, and a `RUN_CODE_FLAVORS` entry in `code-mode.ts` for
|
||||
* its `run_code` schema strings — plus the renderer function this table points
|
||||
* at. The `satisfies` clause pins this table's key set to that union, which
|
||||
* the flavor table is checked against too, so any of the three left out is a
|
||||
* typecheck failure. What no check reaches is the prose that names the values
|
||||
* instead of deriving them: the seam's `dsh-code-runtime` README pair, its
|
||||
* `CodeRuntime.language` JSDoc, and `docs/core-data-structures/code-runtime.md`
|
||||
* with its zh pair, plus this package's own README pair and the
|
||||
* {@link Config.mode} JSDoc.
|
||||
*/
|
||||
const SDK_RENDERERS: Record<string, (schemas: ToolSdkSchema[]) => string> = {
|
||||
typescript: renderToolsSdk,
|
||||
python: renderToolsSdkPy,
|
||||
} satisfies Record<CodeSdkLanguage, (schemas: ToolSdkSchema[]) => string>
|
||||
|
||||
export {
|
||||
defineTool,
|
||||
@@ -65,6 +88,7 @@ export type { JsonValue } from '@deepseek-ai/dsh-session'
|
||||
|
||||
export { CodeRunFailedError, RUN_CODE_NAME } from './code-mode.ts'
|
||||
export { jsonSchemaToTs, renderToolsSdk } from './ts-types.ts'
|
||||
export { jsonSchemaToPy, renderToolsSdkPy } from './py-types.ts'
|
||||
export { defineContentToolFixture, type ContentToolFixtureOptions } from './testing.ts'
|
||||
|
||||
// The render-intent vocabulary a tool declares via `presentCall`/`presentResult`
|
||||
@@ -593,8 +617,9 @@ export interface Config {
|
||||
/**
|
||||
* Model presentation. `native` (default) sends every visible schema; `code`
|
||||
* sends only `run_code` plus a generated SDK prompt; `both` sends both forms.
|
||||
* Code modes require a TypeScript runtime and fail prompt assembly when it is
|
||||
* absent or mismatched. Under `code`, native names in `toolOrder` are invalid.
|
||||
* Code modes require a `ctx.codeRuntime` whose `language` has a registered
|
||||
* SDK renderer (TypeScript or Python) and fail prompt assembly when it is
|
||||
* absent or has no renderer. Under `code`, native names in `toolOrder` are invalid.
|
||||
*/
|
||||
mode?: ToolPresentationMode
|
||||
/**
|
||||
@@ -757,6 +782,7 @@ export class ToolRegistry extends Service {
|
||||
? undefined
|
||||
: createRunCodeTool(this, {
|
||||
requireRuntime: () => this.requireCodeRuntime(),
|
||||
peekRuntime: () => this.ctx.get('codeRuntime'),
|
||||
maxParallel: resolveMaxParallelSubCalls(config.maxParallelSubCalls),
|
||||
shapeDispatchLog: dispatch => this.shapeDispatchLog(dispatch),
|
||||
})
|
||||
@@ -765,10 +791,21 @@ export class ToolRegistry extends Service {
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tools:sdk',
|
||||
order: SDK_SECTION_ORDER,
|
||||
// Regenerate from the calling scope's visible tools in stable order.
|
||||
// Regenerate from the calling scope's visible tools in stable order,
|
||||
// picking the renderer that matches the loaded runtime's language.
|
||||
// `requireCodeRuntime` already validated the language is in the table,
|
||||
// so the guard below is defense-in-depth against a caller that bypassed
|
||||
// it (impossible under normal composition).
|
||||
text: (context) => {
|
||||
this.requireCodeRuntime()
|
||||
return renderToolsSdk(this.sdkSchemas(context.scope))
|
||||
const runtime = this.requireCodeRuntime()
|
||||
// Own-property read: a language like `toString`/`constructor` would
|
||||
// otherwise resolve an inherited Object.prototype member as a renderer.
|
||||
const render = SDK_RENDERERS[runtime.language]
|
||||
/* v8 ignore next 3 -- requireCodeRuntime rejects an unknown language before this ever runs. */
|
||||
if (!Object.hasOwn(SDK_RENDERERS, runtime.language) || render === undefined) {
|
||||
throw new Error(`dsh-tools: no SDK renderer registered for runtime language ${JSON.stringify(runtime.language)} (known: ${Object.keys(SDK_RENDERERS).map(name => JSON.stringify(name)).join(', ')})`)
|
||||
}
|
||||
return render(this.sdkSchemas(context.scope))
|
||||
},
|
||||
})
|
||||
}
|
||||
@@ -780,11 +817,17 @@ export class ToolRegistry extends Service {
|
||||
*/
|
||||
private wireSchemas(scope?: ScopeKey): ToolProviderResult {
|
||||
const view = this.view(scope)
|
||||
const schemas = [...view.visible.values()].map(definition => this.schemaOf(definition, false))
|
||||
if (this.mode === 'native') {
|
||||
const schemas = [...view.visible.values()].map(definition => this.schemaOf(definition, false))
|
||||
return { schemas, knownNames: [...view.knownNames] }
|
||||
}
|
||||
// Validate the runtime language BEFORE projecting schemas: schemaOf reads
|
||||
// run_code's language-aware description/parameters getters, whose own
|
||||
// flavor-table guard would otherwise surface first. This keeps the
|
||||
// renderer-table rejection the canonical assembly-time error for a
|
||||
// language with no SDK renderer.
|
||||
this.requireCodeRuntime()
|
||||
const schemas = [...view.visible.values()].map(definition => this.schemaOf(definition, false))
|
||||
if (this.mode === 'code') {
|
||||
return {
|
||||
schemas: schemas.filter(schema => schema.name === RUN_CODE_NAME),
|
||||
@@ -801,14 +844,23 @@ export class ToolRegistry extends Service {
|
||||
* behind it — hostage to a code runtime existing even under `mode:
|
||||
* 'native'` (the loop's optional-backend idiom, same as
|
||||
* `sessionPersistence`).
|
||||
*
|
||||
* Assembly and `run_code` execution read separately, so the language is not
|
||||
* bound to a request. Harmless while one published backend exists — both
|
||||
* reads return the same flavor — but a reload that swapped in a second
|
||||
* language between them would hand a program written against one SDK to the
|
||||
* other. Binding it belongs to the PR that publishes that backend, which is
|
||||
* also the first point it can be tested; recorded in the
|
||||
* [language-dispatch note](../../../../.agents/notes/implemented/feature/2026-07-31-code-mode-language-dispatch.md).
|
||||
*/
|
||||
private requireCodeRuntime(): CodeRuntime {
|
||||
const runtime = this.ctx.get('codeRuntime')
|
||||
if (!runtime) {
|
||||
throw new Error(`dsh-tools: mode "${this.mode}" requires a code runtime — load a ctx.codeRuntime implementation (e.g. @deepseek-ai/dsh-code-runtime-worker) or set tools mode to "native"`)
|
||||
}
|
||||
if (runtime.language !== 'typescript') {
|
||||
throw new Error(`dsh-tools: mode "${this.mode}" generates a TypeScript SDK, but the loaded code runtime's language is "${runtime.language}"`)
|
||||
if (!Object.hasOwn(SDK_RENDERERS, runtime.language)) {
|
||||
const known = Object.keys(SDK_RENDERERS).map(name => JSON.stringify(name)).join(', ')
|
||||
throw new Error(`dsh-tools: no SDK renderer registered for runtime language ${JSON.stringify(runtime.language)} (known: ${known})`)
|
||||
}
|
||||
return runtime
|
||||
}
|
||||
|
||||
818
packages/core/tools/src/py-types.ts
Normal file
818
packages/core/tools/src/py-types.ts
Normal file
@@ -0,0 +1,818 @@
|
||||
/**
|
||||
* Code Mode codegen — Python flavor. The pure projection from registered tool schemas to the
|
||||
* Python SDK text the model programs against under `runtime.language === 'python'`. Sibling of
|
||||
* {@link ./ts-types.ts | ts-types.ts}; the two files are two projections of the same registry
|
||||
* store, keyed by the loaded {@link @deepseek-ai/dsh-code-runtime#CodeRuntime.language | code
|
||||
* runtime's language}.
|
||||
*
|
||||
* Under `mode: 'code'` the native tool schemas are omitted from the request, so this generated
|
||||
* SDK is the model's ONLY source for each tool's argument names, required fields, types,
|
||||
* descriptions, and canonical output shapes; under `mode: 'both'` the native schemas ship
|
||||
* alongside it and it is one of two. Object-shaped arguments and outputs therefore render as one
|
||||
* named `TypedDict` per tool (and per nested object), not an opaque `dict[str, Any]`, so the
|
||||
* shape survives into the program under the mode that has nothing else to carry it.
|
||||
* @module @deepseek-ai/dsh-tools/src/py-types
|
||||
*/
|
||||
|
||||
import { assertSupportedJsonSchema } from './json-schema.ts'
|
||||
import type { JsonSchemaNode, JsonSchemaScalar } from './json-schema.ts'
|
||||
import type { ToolSdkSchema } from './ts-types.ts'
|
||||
|
||||
/**
|
||||
* The reference grammar's `xid_start xid_continue*` — the set
|
||||
* `str.isidentifier()` accepts on a CPython whose Unicode tables match the
|
||||
* engine's. See {@link isBareIdentifier} for what a version skew does.
|
||||
*/
|
||||
const IDENTIFIER = /^[\p{XID_Start}_]\p{XID_Continue}*$/u
|
||||
|
||||
/**
|
||||
* Whether a name can be emitted as a bare Python identifier rather than
|
||||
* routed to the subscript/`dict[str, Any]` path.
|
||||
*
|
||||
* Python identifiers are not ASCII: `路径` is as legal a field name as `path`,
|
||||
* and rejecting it would degrade the whole enclosing object, dropping every
|
||||
* field's name, requiredness, and type — information whose only source under
|
||||
* `mode: 'code'` is this generated text.
|
||||
*
|
||||
* NFKC stability is a second and separate condition, because CPython
|
||||
* normalizes identifiers at compile time while JSON keys are compared as
|
||||
* written: `field` would be declared and reachable as `field`, so the SDK would
|
||||
* advertise a key under a spelling the harness never accepts, and two keys
|
||||
* that normalize together would collapse into one declaration. Those names
|
||||
* take the subscript path, which carries their exact bytes.
|
||||
*
|
||||
* `IDENTIFIER`'s equivalence to `str.isidentifier()` was measured across 21
|
||||
* samples with zero divergence, on Node 22.23.1 against CPython 3.9.6 — every
|
||||
* sample sits inside the two versions' shared tables, and the skew characters
|
||||
* below are exactly where that pair diverges. The predicate as a whole is
|
||||
* deliberately stricter than `isidentifier()`, which does not test NFKC
|
||||
* stability: `'field'.isidentifier()` is True and this returns false.
|
||||
*
|
||||
* Both conditions are evaluated against the ENGINE's Unicode tables, and the
|
||||
* two sides are versioned independently — `\p{XID_Start}`/`\p{XID_Continue}`
|
||||
* follow the running engine (Node 22.23.1 reports Unicode 17.0) while CPython
|
||||
* follows its own (3.9.6 reports 13.0.0). The skew is not symmetric. A CPython
|
||||
* older than the engine is the dangerous direction: a character added to either
|
||||
* property since its tables (U+10570 Vithkuqi and U+1E290 Toto, 14.0; U+1E4D0
|
||||
* Nag Mundari, 15.0; U+1C89 Cyrillic TJE, 16.0 — ages per `DerivedAge.txt`; all
|
||||
* four are NFKC-stable and accepted here, and all four are `Cn` on that 3.9.6,
|
||||
* which rejects them) is emitted bare and its tokenizer refuses the character,
|
||||
* taking the whole SDK block down — the same parseability invariant
|
||||
* {@link UNPRINTABLE}, {@link LONE_SURROGATE} and {@link MAX_LIST_NESTING}
|
||||
* exist for. Both properties carry it: a character added only to `XID_Continue`
|
||||
* passes the trailing `\p{XID_Continue}*` in a tail position and fails the same
|
||||
* way — U+200C ZWNJ and U+200D ZWJ are that case, gaining `XID_Continue` in UCD
|
||||
* 15.1 and absent from it in 13.0.0, 14.0.0 and 15.0.0, so `a\u{200C}b` is
|
||||
* emitted bare here while `isidentifier()` is False on 3.9.6 and on 3.12.13
|
||||
* (15.0.0). A CPython newer than the engine only routes a legal name to the
|
||||
* subscript/`dict[str, Any]` path: less readable, still correct. The NFKC
|
||||
* condition reduces to the same skew, since normalization stability guarantees
|
||||
* an assigned character's normalization never changes afterwards.
|
||||
*
|
||||
* This predicate is not the only reader of engine tables. {@link camelCase}
|
||||
* reads them at three further points — its split set, its head test, and its
|
||||
* `toUpperCase()` case mapping — and this predicate's verdict gates none of
|
||||
* them: a class name derived there reaches emitted text whenever any object
|
||||
* shape in the tool's schema declares a `TypedDict`, including for a tool this
|
||||
* predicate rejected. A tool named `zz-\u{1E4D0}x` with such parameters never
|
||||
* reaches the skew here (the `-` rejects it outright) yet emits `class
|
||||
* Zz\u{1E4D0}xArgs`, which that same 3.9.6 refuses — Nag Mundari arrived two
|
||||
* releases after its tables. The case mapping is a separate table rather than
|
||||
* an XID membership test, and it fails on names both conditions above accept:
|
||||
* `\u{019B}` is XID_Start and NFKC-stable, so this predicate accepts it and
|
||||
* `async def \u{019B}` compiles on 3.9.6, but Node uppercases it to
|
||||
* `\u{A7DC}` — unassigned in that CPython, whose own `.upper()` is the identity
|
||||
* here — and the declared `class \u{A7DC}Args` fails with `invalid
|
||||
* non-printable character U+A7DC`. Closing the exposure therefore covers all
|
||||
* four read points, not this predicate alone; it needs the target interpreter's
|
||||
* version, which the backend reporting `language: 'python'` owns and which is
|
||||
* unpublished on this base, so the note records it as that PR's decision.
|
||||
*
|
||||
* The `ts-types` sibling keeps its own ASCII rule rather than sharing this
|
||||
* one: ECMAScript identifiers are a different set (`$`) and are never
|
||||
* normalized, so one predicate cannot be correct for both. ZWJ/ZWNJ are not
|
||||
* part of that difference — both sets carry them on the engine's tables; what
|
||||
* separates the two there is the CPython table version above.
|
||||
* @param name - the raw schema field or tool name.
|
||||
* @returns whether the name can be emitted bare.
|
||||
*/
|
||||
function isBareIdentifier(name: string): boolean {
|
||||
return IDENTIFIER.test(name) && name.normalize('NFKC') === name
|
||||
}
|
||||
|
||||
/**
|
||||
* Python hard keywords: reserved everywhere, so a tool or field named
|
||||
* ``class`` or ``lambda`` is legal on the wire but not as an attribute
|
||||
* (``tools.class`` would be a SyntaxError in the model program) and not as a
|
||||
* class-syntax `TypedDict` field. Such a tool renders under subscript access
|
||||
* and such an object degrades to ``dict[str, Any]`` — the model still reaches
|
||||
* every tool and field without collisions.
|
||||
* Soft keywords (``match``, ``case``, ``type``, ``_`` — the language
|
||||
* reference's whole set) are deliberately ABSENT: each is special in exactly
|
||||
* one syntactic position — a statement head (``match``, ``type``), a ``match``
|
||||
* statement's clause head (``case``), or a pattern (``_``) — so ``match: str``
|
||||
* as a field and ``async def match(...)`` as a method are both legal, and
|
||||
* including them would needlessly degrade common search/regex tool fields to
|
||||
* ``dict[str, Any]``. Underscore-leading names are handled separately, not
|
||||
* here: a non-dunder ``__token`` name-mangles, a dunder present on
|
||||
* ``object``/``type`` resolves before the proxy hook, and implicit
|
||||
* special-method lookup bypasses the hook.
|
||||
*/
|
||||
const RESERVED = new Set([
|
||||
'False', 'None', 'True', 'and', 'as', 'assert', 'async', 'await', 'break', 'class',
|
||||
'continue', 'def', 'del', 'elif', 'else', 'except', 'finally', 'for', 'from', 'global',
|
||||
'if', 'import', 'in', 'is', 'lambda', 'nonlocal', 'not', 'or', 'pass', 'raise',
|
||||
'return', 'try', 'while', 'with', 'yield',
|
||||
// Not a keyword, but CPython refuses to ASSIGN it at compile time
|
||||
// (`SyntaxError: cannot assign to __debug__`), which is what a TypedDict
|
||||
// field, a parameter name, and a keyword argument all are.
|
||||
'__debug__',
|
||||
])
|
||||
|
||||
/** `typing` symbols this module may emit, in the deterministic import order. */
|
||||
const TYPING_ORDER = ['Any', 'Literal', 'NotRequired', 'Protocol', 'TypedDict'] as const
|
||||
|
||||
/** `indent`-deep line prefix (four spaces per level to match PEP 8 output). */
|
||||
function pad(indent: number): string {
|
||||
return ' '.repeat(indent)
|
||||
}
|
||||
|
||||
/**
|
||||
* Collector threaded through {@link renderType}: the emitted `TypedDict` class
|
||||
* declarations (nested classes precede the parent that references them), the
|
||||
* class names already taken (for collision suffixing), a per-base collision
|
||||
* counter, and the `typing` symbols the render actually used.
|
||||
*/
|
||||
interface RenderState {
|
||||
readonly classes: string[]
|
||||
readonly usedClassNames: Set<string>
|
||||
/** Next collision counter per capped base, so allocation is amortized O(1) instead of rescanning from `2`. */
|
||||
readonly nextClassCounter: Map<string, number>
|
||||
readonly typing: Set<string>
|
||||
}
|
||||
|
||||
/**
|
||||
* The `Cc` code points that survive the whitespace collapse in {@link describe}
|
||||
* and have no printable form: the C0 controls, DEL, and the C1 controls. Only
|
||||
* U+0009 to U+000D are absent, because ECMAScript `\s` already collapsed them —
|
||||
* `\s` is TAB/VT/FF/SP/NBSP/ZWNBSP/Zs plus LF/CR/LS/PS, so no C1 code point is
|
||||
* in it and the whole U+0080 to U+009F block reaches this rule intact. Those
|
||||
* are not hypothetical input: they are what Windows-1252 bytes 0x80 to 0x9F
|
||||
* (smart quotes, em dash) become when decoded as Latin-1.
|
||||
* CPython rejects source containing a NUL outright
|
||||
* (`SyntaxError: source code string cannot contain null bytes`), whether it
|
||||
* sits in a docstring or in a comment, so one such byte anywhere in a schema
|
||||
* description would make the whole generated SDK unparseable — under
|
||||
* `mode: 'code'`, the model's only declaration of the tools. The rest are
|
||||
* legal but invisible; escaping them with the same rule keeps the emitted text
|
||||
* readable and the treatment uniform.
|
||||
*
|
||||
* The boundary is the category, not per-code-point addressability: `\xNN`
|
||||
* addresses U+0000 to U+00FF, so one escape form covers `Cc` exactly. The
|
||||
* invisible `Cf` formatting characters pass through by design — of them only
|
||||
* U+00AD soft hyphen would fit `\xNN` at all, and escaping that one while
|
||||
* U+200B ZWSP, U+200E/U+200F bidi marks, and U+2060 word joiner passed through
|
||||
* would leave a rule that is neither category- nor addressability-shaped. The
|
||||
* whole family is legal in both consumers, since only LF and CR terminate a
|
||||
* Python string literal or a `#` comment. That set is the tokenizer's, not
|
||||
* `str.splitlines()`': NEL (U+0085), LS (U+2028), and PS (U+2029) split a
|
||||
* string at run time but do not end a physical line in source — measured on
|
||||
* CPython 3.9.6 and 3.12.13, each accepted in both positions with the value
|
||||
* round-tripping — so they are safe raw wherever they reach emitted text
|
||||
* unescaped, which for all three is `JSON.stringify`, at two call sites:
|
||||
* {@link pyScalar}'s literal path, and the subscript tool-name comment's own
|
||||
* call, which a name carrying any of them always reaches, none being
|
||||
* `XID_Continue`. The `description` path escapes NEL under the class above and
|
||||
* folds LS and PS in {@link describe}'s `\s+` collapse, both being `\s`.
|
||||
*/
|
||||
const UNPRINTABLE = /[\u0000-\u0008\u000e-\u001f\u007f-\u009f]/g
|
||||
|
||||
/**
|
||||
* Unpaired surrogate code points, escaped by {@link describe} as `\uNNNN` —
|
||||
* its own form, since `\xNN` stops at U+00FF. The `u` flag is what makes this
|
||||
* the LONE ones: in Unicode mode a well-formed pair is a single astral code
|
||||
* point outside D800 to DFFF, so an emoji in a description survives untouched.
|
||||
*
|
||||
* This is the NUL case from {@link UNPRINTABLE}, not the invisible-character
|
||||
* case. Python source must be UTF-8-encodable and a lone surrogate is not, so
|
||||
* `compile()` raises `UnicodeEncodeError: surrogates not allowed` for one
|
||||
* anywhere in the text — measured on 3.9 for a string literal and for a `#`
|
||||
* comment alike. A raw or MCP tool description reaches this: `JSON.parse` on a
|
||||
* wire `"\ud800"` escape yields exactly such a code point.
|
||||
*/
|
||||
const LONE_SURROGATE = /[\ud800-\udfff]/gu
|
||||
|
||||
/**
|
||||
* The collapsed one-line `description` of a schema node (byte-stable across
|
||||
* formatting churn), or `undefined` when the node carries none. Every caller
|
||||
* passes an object — a validated property node, the `ToolSdkSchema` itself, or
|
||||
* the `{ description }` wrapper {@link docLines} synthesizes — so only the
|
||||
* description field needs guarding. A description that collapses
|
||||
* to nothing (empty, or whitespace only) is `undefined` too: it documents the
|
||||
* node no better than an absent one, and emitting it would leave an empty
|
||||
* `"""` docstring or a bare `# ` line in the SDK. Only ECMAScript whitespace
|
||||
* folds, so a description of whitespace plus one surviving control character is
|
||||
* NOT absent: it collapses to that character's visible escape.
|
||||
*
|
||||
* Control characters left over after the whitespace collapse are rendered as
|
||||
* their `\xNN` escapes (see {@link UNPRINTABLE}) and unpaired surrogates as
|
||||
* their `\uNNNN` escapes (see {@link LONE_SURROGATE}); the escape's own backslash is
|
||||
* emitted literally by both consumers, since {@link docLines} doubles it into a
|
||||
* Python source escape and a `#` comment carries it verbatim.
|
||||
*/
|
||||
function describe(schema: object): string | undefined {
|
||||
const description = (schema as Record<string, unknown>).description
|
||||
if (typeof description !== 'string') return undefined
|
||||
const collapsed = description
|
||||
.replace(/\s+/g, ' ')
|
||||
.replace(UNPRINTABLE, char => `\\x${char.charCodeAt(0).toString(16).padStart(2, '0')}`)
|
||||
.replace(LONE_SURROGATE, char => `\\u${char.charCodeAt(0).toString(16).padStart(4, '0')}`)
|
||||
.trim()
|
||||
return collapsed.length === 0 ? undefined : collapsed
|
||||
}
|
||||
|
||||
/**
|
||||
* One-line docstring for a tool `description`, or no lines when there is none.
|
||||
* Backslashes are doubled first, every quote is escaped, and a trailing
|
||||
* backslash cannot survive: a description ending in `"` or an odd backslash
|
||||
* would otherwise merge with (or escape) the closing triple quote and make
|
||||
* the generated block — Code Mode's only SDK — syntactically invalid Python.
|
||||
*/
|
||||
function docLines(description: unknown, indent: number): string[] {
|
||||
const collapsed = describe({ description })
|
||||
if (collapsed === undefined) return []
|
||||
const escaped = collapsed.replaceAll('\\', '\\\\').replaceAll('"', '\\"')
|
||||
return [`${pad(indent)}"""${escaped}"""`]
|
||||
}
|
||||
|
||||
/**
|
||||
* CamelCase a name into a Python type identifier: non-identifier characters
|
||||
* split words, `_` splits too (it is `XID_Continue`, so the split set names it
|
||||
* explicitly), and a head that cannot start an identifier takes a `Tool`
|
||||
* prefix. Unicode survives, so a `路径` field yields `路径`-based class names
|
||||
* instead of collapsing to the bare prefix. A character that is not
|
||||
* `XID_Continue` splits even when it is a letter, so a name whose NFKC folding
|
||||
* would leave the identifier set is not carried through — the split set is the
|
||||
* grammar's, not an ASCII approximation of it.
|
||||
*
|
||||
* The result is NFKC-normalized: these names are generated, never matched
|
||||
* against a JSON key, so normalizing is free here and keeps what CPython
|
||||
* compiles identical to what is emitted — unlike {@link isBareIdentifier},
|
||||
* which must reject unstable names outright. Normalizing AFTER the prefix
|
||||
* decision is what makes that hold at the seam the prefix creates: `Tool` +
|
||||
* a combining-mark head composes there (`U+0301` gives `Tooĺ`, U+013A), so
|
||||
* normalizing only the un-prefixed part would emit a name CPython compiles to
|
||||
* a different symbol. The second call is idempotent on the un-prefixed arm.
|
||||
*
|
||||
* The split set, the head test, and `toUpperCase()` all read the engine's
|
||||
* Unicode tables, so this function carries the same version skew
|
||||
* {@link isBareIdentifier} documents, by paths independent of it: a class name
|
||||
* derived here reaches emitted text whenever any object shape in the tool's
|
||||
* schema declares a `TypedDict`, and the predicate's verdict on the tool name
|
||||
* does not gate that. The case mapping is the one that can fail on a name the
|
||||
* predicate accepted; the worked example is there.
|
||||
* @param raw - the schema field or tool name to derive from.
|
||||
* @returns a class-name segment safe to emit.
|
||||
*/
|
||||
function camelCase(raw: string): string {
|
||||
const joined = raw
|
||||
.split(/[^\p{XID_Continue}]+|_+/u)
|
||||
.filter(part => part.length > 0)
|
||||
.map(part => `${part.charAt(0).toUpperCase()}${part.slice(1)}`)
|
||||
.join('')
|
||||
.normalize('NFKC')
|
||||
return (/^\p{XID_Start}/u.test(joined) ? joined : `Tool${joined}`).normalize('NFKC')
|
||||
}
|
||||
|
||||
/** Class-name base cap keeping each emitted name — and total text — linear in schema depth. */
|
||||
const MAX_CLASS_NAME_BASE = 120
|
||||
|
||||
/**
|
||||
* Deepest `list[…]` nesting emitted into one annotation before the item type
|
||||
* degrades to `Any`. CPython's tokenizer rejects a logical line holding more
|
||||
* than 200 simultaneously-open brackets (`MAXLEVEL`, `SyntaxError: too many
|
||||
* nested parentheses`), so an array chain deeper than that would render an SDK
|
||||
* block that is not valid Python at all — the same failure the docstring
|
||||
* escaping in {@link docLines} exists to prevent. 180 leaves headroom for the
|
||||
* few brackets an annotation can add around the chain, all of which count
|
||||
* toward the same limit. Per emission site, counting brackets open at the
|
||||
* chain's innermost point:
|
||||
*
|
||||
* - Return annotation, `async def f(self, args: X) -> chain:` — 180 `list[`
|
||||
* plus an innermost `Literal[`. The parameter list's `(` closed at the `)`
|
||||
* before the `->`, so it is NOT open here: 181.
|
||||
* - TypedDict field, `field: NotRequired[chain]` — a class-body line with no
|
||||
* other open bracket, and its children start at `listDepth: 1` to reserve
|
||||
* the `NotRequired[`, so 179 `list[` plus `Literal[`: 181. Required fields
|
||||
* share that start for uniformity, spending one level of representable depth
|
||||
* on a bracket they never emit.
|
||||
* - Argument annotation, `async def f(self, args: chain) -> Y:` — the `(` IS
|
||||
* still open around it: 180 `list[` plus `Literal[` plus the paren, 182, the
|
||||
* worst case. Reachable only through a raw `register()` whose `parameters`
|
||||
* is an array reached from the root through `oneOf` arms alone — the root
|
||||
* array itself, or one nested under any depth of unions, since an arm
|
||||
* inherits the enclosing depth unchanged (`A | B` opens no bracket). An
|
||||
* object ancestor takes it out of this case: its fields restart the chain at
|
||||
* the 181 site. `defineTool` compiles an object root, so the annotation is a
|
||||
* bare TypedDict class name or a one-bracket `dict[str, Any]` when that
|
||||
* object degrades — never a chain.
|
||||
*
|
||||
* A CPython grammar limit, not a deployment choice, so it is fixed rather than
|
||||
* configurable. The sibling `ts-types` renderer needs no counterpart: nothing
|
||||
* in the TypeScript grammar bounds nesting, and its SDK block is never type-
|
||||
* checked. Only bracket nesting counts — a `oneOf` renders as a flat `A | B`
|
||||
* chain and nested objects render as separate `class` statements, so neither
|
||||
* accumulates open brackets at any depth. The invariant this cap serves is
|
||||
* grammatical validity; see the `oneOf` arm in {@link renderType} for the one
|
||||
* interpreter limit deliberately left uncapped.
|
||||
*/
|
||||
const MAX_LIST_NESTING = 180
|
||||
|
||||
/**
|
||||
* Cap a class-name base at {@link MAX_CLASS_NAME_BASE} (see the callers for
|
||||
* why capping keeps the render linear). `slice` counts UTF-16 code units, so
|
||||
* an astral character straddling the boundary would be cut in half and leave a
|
||||
* lone surrogate — not an identifier character, and not even well-formed text;
|
||||
* drop it rather than emit it.
|
||||
*/
|
||||
function capClassNameBase(base: string): string {
|
||||
if (base.length <= MAX_CLASS_NAME_BASE) return base
|
||||
const capped = base.slice(0, MAX_CLASS_NAME_BASE)
|
||||
return /[\uD800-\uDBFF]$/.test(capped) ? capped.slice(0, -1) : capped
|
||||
}
|
||||
|
||||
/**
|
||||
* Reserve a unique class name from a base, suffixing `2`, `3`, … on collision.
|
||||
* The base is capped at {@link MAX_CLASS_NAME_BASE} first: child class names
|
||||
* derive from their parent's allocated name (`ParentChild`), so an unbounded
|
||||
* schema of single-field objects would otherwise grow each name by one field
|
||||
* per level and the sum of all names to Θ(depth²). Capping the base keeps each
|
||||
* name — and the total emitted text — linear in depth. Collisions resume from
|
||||
* the per-base counter in `state.nextClassCounter` rather than rescanning from
|
||||
* `2`, so a deep chain sharing one capped base stays O(1) per allocation
|
||||
* (amortized) instead of Θ(depth²) in time.
|
||||
*/
|
||||
function allocateClassName(base: string, state: RenderState): string {
|
||||
const capped = capClassNameBase(base)
|
||||
let name = capped
|
||||
if (state.usedClassNames.has(name)) {
|
||||
let n = state.nextClassCounter.get(capped) ?? 2
|
||||
while (state.usedClassNames.has(`${capped}${n}`)) n++
|
||||
name = `${capped}${n}`
|
||||
state.nextClassCounter.set(capped, n + 1)
|
||||
}
|
||||
state.usedClassNames.add(name)
|
||||
return name
|
||||
}
|
||||
|
||||
/**
|
||||
* Append a child-name segment to a parent class-name base, capping the result
|
||||
* at {@link MAX_CLASS_NAME_BASE}. Capping AT PROPAGATION (not only inside
|
||||
* {@link allocateClassName}) keeps each level O(1): a deep `oneOf`- or
|
||||
* object-chain would otherwise carry an ever-growing ConsString down the tree
|
||||
* and re-materialize it (via `.length`/`.slice`) at every level — Θ(depth²).
|
||||
* The bounded base plus the collision counter still yields unique names.
|
||||
*
|
||||
* The join is NFKC-normalized because both sides are separately normalized yet
|
||||
* their concatenation need not be: a base ending in a Hangul L jamo or LV
|
||||
* syllable composes with a following V or T jamo head (`가` + `ᆨ` gives `각`),
|
||||
* so the emitted class name would differ from the symbol CPython compiles, and
|
||||
* two byte-distinct names could fold onto one — `usedClassNames` dedupes by the
|
||||
* raw bytes, so the collision counter would not see it. Normalizing costs
|
||||
* O(cap + segment) per level, the same order as the `slice` it feeds. The other
|
||||
* two join points need no counterpart: `Args`/`Output` start with `A`/`O` and
|
||||
* {@link allocateClassName}'s suffix is digits, none of which compose backwards.
|
||||
*/
|
||||
function childClassName(base: string, segment: string): string {
|
||||
return capClassNameBase(`${base}${segment}`.normalize('NFKC'))
|
||||
}
|
||||
|
||||
/**
|
||||
* Render one validated scalar as Python literal text (`True`/`False`,
|
||||
* JSON-quoted strings, bare numbers). `null` cannot reach here: the `null`
|
||||
* type renders directly as `None`, and the unified validator rejects a null
|
||||
* `const`/`enum` entry on every other scalar type.
|
||||
*
|
||||
* A beyond-safe-range integral number takes `BigInt` digits rather than
|
||||
* `String`: Python integers are arbitrary-precision, so the emitted digits ARE
|
||||
* the value the model programs against, and `String` can give a different
|
||||
* integer than the double holds (`2 ** 60` prints the rounded `...847000`, not
|
||||
* the exact `...846976`) or no integer literal at all (`1e21` prints `1e+21`).
|
||||
* `String`'s rounding is not a bug in it: `Number::toString` emits the shortest
|
||||
* decimal string that re-reads to the same double, then pads to the exponent
|
||||
* with zeros (1 significant digit for `1e20`, 16 for `2 ** 60`) — and when the
|
||||
* shortest string is shorter than the double's exact value, those padded digits
|
||||
* name an integer no double holds. Passing one back would have to cross the
|
||||
* argument boundary as a JSON number — a double again — so the SDK would
|
||||
* document a value no program can pass. `BigInt` needs no case split: where
|
||||
* `String` is already exact (`2 ** 53`, `1e20`) the two agree byte for byte,
|
||||
* and where it is not, `BigInt` is the exact one. The TS flavor needs no
|
||||
* counterpart at all: its literal is re-read by a JS parser back into the same
|
||||
* double.
|
||||
*
|
||||
* `JSON.stringify` is also what keeps this path's output parseable, and it is
|
||||
* the only thing that does. It covers both classes of hazard: the two kinds of
|
||||
* code point CPython refuses anywhere in source — NUL among the C0 controls,
|
||||
* and the whole D800–DFFF unpaired-surrogate block, escaped under ES2019
|
||||
* well-formed stringification, which the engines range guarantees — and the
|
||||
* ones that break this line in particular, a bare `"` closing the literal
|
||||
* early, a trailing odd backslash eating the closing quote, and a bare LF/CR
|
||||
* ending it before its terminator. The `description` path carries
|
||||
* {@link UNPRINTABLE} and {@link LONE_SURROGATE} because nothing quotes it,
|
||||
* and folds newlines in {@link describe}.
|
||||
*
|
||||
* That leans on a coincidence worth naming: every escape `JSON.stringify` can
|
||||
* emit (`\"`, `\\`, `\b`, `\f`, `\n`, `\r`, `\t`, `\uXXXX`) is also a Python
|
||||
* escape denoting the same character, so the emitted `Literal[...]` both
|
||||
* parses and decodes back to the value the schema declared. DEL, the C1
|
||||
* controls (NEL among them), and LS/PS (U+2028/U+2029) do reach it raw —
|
||||
* legal but invisible, byte-for-byte as in the TS flavor; escaping them is a
|
||||
* both-flavors change. Those last three are legal here for the reason
|
||||
* {@link UNPRINTABLE} records: they are `str.splitlines()` boundaries, not
|
||||
* tokenizer line terminators. The subscript tool-name comment quotes its name
|
||||
* through its own call to the same `JSON.stringify`, never through this
|
||||
* function, and inherits both halves — escapes and pass-throughs alike.
|
||||
*/
|
||||
function pyScalar(value: JsonSchemaScalar): string {
|
||||
if (value === true) return 'True'
|
||||
if (value === false) return 'False'
|
||||
if (typeof value === 'string') return JSON.stringify(value)
|
||||
if (typeof value === 'number' && Number.isInteger(value) && !Number.isSafeInteger(value)) {
|
||||
return BigInt(value).toString()
|
||||
}
|
||||
return String(value)
|
||||
}
|
||||
|
||||
/**
|
||||
* Render a validated scalar `const`/`enum` as `Literal[...]`, falling back to
|
||||
* the broad type. Deliberately deviates from PEP 586, which restricts `Literal`
|
||||
* parameters to int/bool/str/bytes/enum/None: a non-integral number
|
||||
* `const`/`enum` emits a float literal (`Literal[1.5]`) a strict checker would
|
||||
* reject. An integral one does not deviate — {@link pyScalar} emits int digits,
|
||||
* including for the beyond-safe-range values it widens through `BigInt`, and
|
||||
* PEP 586 admits int parameters. Harmless either way — the stub is advisory
|
||||
* prompt text, only required to parse — and keeping the exact value
|
||||
* communicates the constraint to the model.
|
||||
*/
|
||||
function renderConstrainedScalar(node: JsonSchemaNode, broad: string, state: RenderState): string {
|
||||
if (node.const !== undefined) {
|
||||
state.typing.add('Literal')
|
||||
return `Literal[${pyScalar(node.const)}]`
|
||||
}
|
||||
if (node.enum !== undefined) {
|
||||
state.typing.add('Literal')
|
||||
return `Literal[${node.enum.map(pyScalar).join(', ')}]`
|
||||
}
|
||||
return broad
|
||||
}
|
||||
|
||||
/**
|
||||
* Map one JSON-Schema node to a Python type expression, threading `state` to
|
||||
* collect the `TypedDict` declarations and `typing` symbols a full render
|
||||
* needs. `className` is the name to give an object node with properties (and
|
||||
* the prefix for its nested objects). Handles every unified schema construct —
|
||||
* `oneOf` (→ `X | Y`), `const`/`enum` (→ `Literal[...]`), `integer` (→ `int`),
|
||||
* `null` (→ `None`) — and degrades an unsupported or malformed schema to `Any`
|
||||
* without throwing, the same trusted-after-validation stance as the sibling
|
||||
* {@link ./ts-types.ts | ts-types} renderer. {@link jsonSchemaToPy} is the
|
||||
* context-free entry point; this is the collecting core.
|
||||
*/
|
||||
function renderType(schema: unknown, className: string, state: RenderState): string {
|
||||
interface Frame {
|
||||
// A validated JSON-schema node past the root `assertSupportedJsonSchema`
|
||||
// (the root frame's schema is asserted before any frame is built), so the
|
||||
// walk reads its fields without casts — the same typed-frame shape as the
|
||||
// sibling ts-types renderer.
|
||||
schema: JsonSchemaNode
|
||||
className: string
|
||||
phase: 'start' | 'children'
|
||||
kind?: 'oneOf' | 'array' | 'typeddict'
|
||||
node?: JsonSchemaNode
|
||||
/** Open `list[` brackets enclosing this node in the annotation being built ({@link MAX_LIST_NESTING}). */
|
||||
listDepth: number
|
||||
children: { schema: JsonSchemaNode; className: string; listDepth: number }[]
|
||||
childIndex: number
|
||||
childTypes: string[]
|
||||
entries: [string, JsonSchemaNode][]
|
||||
allocated?: string
|
||||
}
|
||||
const newFrame = (schema: JsonSchemaNode, className: string, listDepth: number): Frame =>
|
||||
({ schema, className, phase: 'start', listDepth, children: [], childIndex: 0, childTypes: [], entries: [] })
|
||||
try {
|
||||
// Validate the WHOLE tree once, then trust it — the same contract the
|
||||
// sibling ts-types renderer follows at a typed same-process seam. Every
|
||||
// node past this point is a validated JSON-schema node, so the walk reads
|
||||
// its fields without re-checking. An unsupported or malformed schema throws
|
||||
// here (before anything is emitted) and degrades to `Any`, the Python
|
||||
// counterpart of the TS flavor's `unknown`.
|
||||
assertSupportedJsonSchema(schema)
|
||||
const frames: Frame[] = [newFrame(schema, className, 0)]
|
||||
let result: string | undefined
|
||||
/* jscpd:ignore-start -- the explicit-stack walk skeleton deliberately parallels
|
||||
ts-types.ts's renderSupportedSchema; the two sibling renderers keep symmetric shapes. */
|
||||
const finish = (type: string): void => {
|
||||
frames.pop()
|
||||
const parent = frames.at(-1)
|
||||
if (parent === undefined) result = type
|
||||
else parent.childTypes.push(type)
|
||||
}
|
||||
|
||||
while (frames.length > 0) {
|
||||
const frame = frames.at(-1)
|
||||
/* v8 ignore next -- the loop condition guarantees a current frame. */
|
||||
if (frame === undefined) break
|
||||
|
||||
if (frame.phase === 'children') {
|
||||
if (frame.childIndex < frame.children.length) {
|
||||
const child = frame.children[frame.childIndex]
|
||||
/* v8 ignore next -- childIndex is bounded by children.length. */
|
||||
if (child === undefined) throw new Error('missing python render child')
|
||||
frame.childIndex++
|
||||
frames.push(newFrame(child.schema, child.className, child.listDepth))
|
||||
continue
|
||||
}
|
||||
if (frame.kind === 'oneOf') {
|
||||
// Concatenate incrementally (template literal, not `Array.join`): V8
|
||||
// builds a lazy ConsString, so a deep oneOf chain materializes once
|
||||
// at the root instead of re-materializing the accumulated string at
|
||||
// every level (which `join` would, making it Θ(depth²)). This matches
|
||||
// the array arm's template-literal laziness and ts-types' composable-
|
||||
// document approach — the whole walk stays linear in schema depth.
|
||||
let union = ''
|
||||
for (const [index, childType] of frame.childTypes.entries()) {
|
||||
union = index === 0 ? childType : `${union} | ${childType}`
|
||||
}
|
||||
finish(union)
|
||||
continue
|
||||
}
|
||||
/* jscpd:ignore-end */
|
||||
if (frame.kind === 'array') {
|
||||
// `list[A | B]` needs no parentheses in Python. Array frames always
|
||||
// schedule exactly one child, so its type is present.
|
||||
/* v8 ignore next -- the ?? arm needs a childless array frame, which start never builds. */
|
||||
finish(`list[${frame.childTypes[0] ?? 'Any'}]`)
|
||||
continue
|
||||
}
|
||||
// typeddict: assemble AFTER the children so any nested class this one
|
||||
// references is already declared (declaration order = reference order).
|
||||
const node = frame.node
|
||||
const name = frame.allocated
|
||||
/* v8 ignore next -- typeddict frames always set node and allocated at start. */
|
||||
if (node === undefined || name === undefined) throw new Error('missing typeddict frame state')
|
||||
const required = new Set(node.required)
|
||||
const lines = [`class ${name}(TypedDict):`]
|
||||
for (let index = 0; index < frame.entries.length; index++) {
|
||||
const entry = frame.entries[index]
|
||||
const fieldType = frame.childTypes[index]
|
||||
/* v8 ignore next -- entries and childTypes correspond one-to-one. */
|
||||
if (entry === undefined || fieldType === undefined) throw new Error('missing typeddict field type')
|
||||
const [field, fieldSchema] = entry
|
||||
// The parent node passed assertSupportedJsonSchema, so every property
|
||||
// value is a validated schema node.
|
||||
const description = describe(fieldSchema)
|
||||
if (description !== undefined) lines.push(`${pad(1)}# ${description}`)
|
||||
if (required.has(field)) {
|
||||
lines.push(`${pad(1)}${field}: ${fieldType}`)
|
||||
} else {
|
||||
state.typing.add('NotRequired')
|
||||
lines.push(`${pad(1)}${field}: NotRequired[${fieldType}]`)
|
||||
}
|
||||
}
|
||||
// TypedDict syntax cannot express openness, so an open object states it
|
||||
// in-band: the annotation is advisory either way, and `mode: 'code'`
|
||||
// omits the native schemas, making this line the model's only signal
|
||||
// that extra keys are accepted.
|
||||
if (node.additionalProperties !== false) {
|
||||
lines.push(`${pad(1)}# Additional keys beyond those declared are allowed.`)
|
||||
}
|
||||
// A closed empty object still needs a class body (`pass`) to be valid
|
||||
// Python; the declared emptiness is the information.
|
||||
if (lines.length === 1) lines.push(`${pad(1)}pass`)
|
||||
state.classes.push(lines.join('\n'))
|
||||
finish(name)
|
||||
continue
|
||||
}
|
||||
|
||||
frame.phase = 'children'
|
||||
const node = frame.schema
|
||||
if (node.oneOf !== undefined) {
|
||||
frame.kind = 'oneOf'
|
||||
// A union renders as `A | B` — no brackets of its own, so the branches
|
||||
// inherit the enclosing depth unchanged.
|
||||
//
|
||||
// Union LENGTH is deliberately uncapped, unlike list nesting. The two
|
||||
// limits are different in kind: >200 open brackets is a SyntaxError
|
||||
// from the tokenizer, so the text is not Python; a long `A | B | …`
|
||||
// chain is grammatically valid at any length and only defeats CPython's
|
||||
// C-recursion when `compile()` walks the left-nested BinOp spine
|
||||
// (measured: 1,000 branches compile, 5,000 raise RecursionError). This
|
||||
// block is prompt text — nothing compiles it — so that limit costs
|
||||
// nothing here, while capping would retire the deep-chain tests that
|
||||
// pin the walk's linear time and the class-name propagation cap. The
|
||||
// standard this renderer holds is grammatical validity, not
|
||||
// compilability under one interpreter's stack.
|
||||
frame.children = node.oneOf.map((branch, index) => ({ schema: branch, className: childClassName(frame.className, `${index + 1}`), listDepth: frame.listDepth }))
|
||||
continue
|
||||
}
|
||||
if (node.type === undefined) {
|
||||
state.typing.add('Any')
|
||||
finish('Any')
|
||||
continue
|
||||
}
|
||||
switch (node.type) {
|
||||
case 'string': finish(renderConstrainedScalar(node, 'str', state)); break
|
||||
case 'number': finish(renderConstrainedScalar(node, 'float', state)); break
|
||||
case 'integer': finish(renderConstrainedScalar(node, 'int', state)); break
|
||||
case 'boolean': finish(renderConstrainedScalar(node, 'bool', state)); break
|
||||
case 'null': finish('None'); break
|
||||
case 'array': {
|
||||
if (node.items === undefined) {
|
||||
state.typing.add('Any')
|
||||
finish('list[Any]')
|
||||
break
|
||||
}
|
||||
// Past MAX_LIST_NESTING another `list[` would push the annotation
|
||||
// beyond CPython's open-bracket limit and make the whole SDK block
|
||||
// unparseable, so the chain degrades here instead — an unusable
|
||||
// annotation either way, and this one is valid Python.
|
||||
if (frame.listDepth >= MAX_LIST_NESTING) {
|
||||
state.typing.add('Any')
|
||||
finish('Any')
|
||||
break
|
||||
}
|
||||
// An array of objects names its item type after the array field.
|
||||
frame.kind = 'array'
|
||||
frame.children = [{ schema: node.items, className: frame.className, listDepth: frame.listDepth + 1 }]
|
||||
break
|
||||
}
|
||||
case 'object': {
|
||||
// A missing `properties` is an empty property map, exactly as the
|
||||
// unified validator and the TS renderer read it — NOT an unknown
|
||||
// shape. The openness of the resulting empty object is decided below,
|
||||
// so a closed empty object still declares an empty TypedDict rather
|
||||
// than a permissive `dict[str, Any]`.
|
||||
const entries = Object.entries(node.properties ?? {})
|
||||
// An empty `className` marks the context-free `jsonSchemaToPy` entry:
|
||||
// there is no naming context to declare into, so degrade. This reads
|
||||
// the CALL's className, not `frame.className`: the marker belongs to
|
||||
// the whole walk, and frames propagate a derived name (a `oneOf`
|
||||
// branch of the context-free root gets the index-derived name `1` —
|
||||
// `childClassName` concatenates and caps, it does not go through
|
||||
// `camelCase`), so a per-frame read would declare classes the caller
|
||||
// has no way to receive, under a name that is not even a legal
|
||||
// identifier: `class 1(TypedDict):`. A field
|
||||
// name that is not a legal Python attribute is inexpressible as a
|
||||
// class-syntax `TypedDict` field, so such an object degrades whole.
|
||||
// A leading-double-underscore non-dunder field (`__token`) would be
|
||||
// NAME-MANGLED inside class syntax (`_ClassName__token`), describing a
|
||||
// different JSON key than the registered schema — degrade like any
|
||||
// other inexpressible field name.
|
||||
if (className === '' || !entries.every(([name]) => isBareIdentifier(name) && !RESERVED.has(name) && !(name.startsWith('__') && !name.endsWith('__')))) {
|
||||
state.typing.add('Any')
|
||||
finish('dict[str, Any]')
|
||||
break
|
||||
}
|
||||
// An OPEN empty object is any dict; a CLOSED empty object declares an
|
||||
// empty TypedDict so "no keys accepted" survives into the SDK.
|
||||
if (entries.length === 0 && node.additionalProperties !== false) {
|
||||
state.typing.add('Any')
|
||||
finish('dict[str, Any]')
|
||||
break
|
||||
}
|
||||
frame.kind = 'typeddict'
|
||||
frame.node = node
|
||||
frame.allocated = allocateClassName(frame.className, state)
|
||||
state.typing.add('TypedDict')
|
||||
frame.entries = entries
|
||||
// A field annotation is its own logical line, so nesting restarts —
|
||||
// at 1, reserving the bracket an optional field's `NotRequired[…]`
|
||||
// wraps around it. frame.allocated was assigned three statements up;
|
||||
// the ?? arm is for the type system only.
|
||||
/* v8 ignore next -- allocated is always set before children are built. */
|
||||
frame.children = entries.map(([field, child]) => ({ schema: child, className: childClassName(frame.allocated ?? '', camelCase(field)), listDepth: 1 }))
|
||||
break
|
||||
}
|
||||
/* v8 ignore next 4 -- assertSupportedJsonSchema narrowed this closed type union. */
|
||||
default: {
|
||||
state.typing.add('Any')
|
||||
finish('Any')
|
||||
}
|
||||
}
|
||||
}
|
||||
/* v8 ignore next -- every root frame produces one expression. */
|
||||
return result ?? 'Any'
|
||||
} catch {
|
||||
// An unsupported or malformed schema failed validation (before any
|
||||
// emission), or an unreachable internal invariant tripped. Either degrades
|
||||
// the node to `Any` rather than crashing prompt assembly — the Python
|
||||
// counterpart of the TS flavor's `unknown` fallback.
|
||||
state.typing.add('Any')
|
||||
return 'Any'
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Map one JSON-Schema node to a context-free Python type expression from the
|
||||
* `typing` module. Handles every unified schema construct — `object` (degraded
|
||||
* to `dict[str, Any]`: naming a `TypedDict` requires the render context that
|
||||
* {@link renderToolsSdkPy} supplies), `const`/`enum` (→ `Literal[...]`),
|
||||
* `oneOf` (→ union), `string`/`number`/`integer`/`boolean`/`null`, `array`
|
||||
* (`items` → `list[T]`) — and returns `Any` for an unsupported or malformed
|
||||
* schema, matching the TS flavor's `unknown` fallback. Type annotations in the
|
||||
* emitted SDK are advisory: Python does not enforce them at runtime.
|
||||
* @param schema - the JSON-Schema node.
|
||||
* @returns the Python type text.
|
||||
*/
|
||||
export function jsonSchemaToPy(schema: unknown): string {
|
||||
// A throwaway state whose class collector never escapes: an object with
|
||||
// properties has nowhere to declare its TypedDict and degrades to
|
||||
// dict[str, Any]. renderToolsSdkPy drives the named-TypedDict path.
|
||||
return renderType(schema, '', { classes: [], usedClassNames: new Set(), nextClassCounter: new Map(), typing: new Set() })
|
||||
}
|
||||
|
||||
/** The fixed model-facing usage contract rendered above the declarations. */
|
||||
const SDK_INSTRUCTIONS = `## Writing code for run_code
|
||||
|
||||
Pass \`run_code\` the body of an async Python function (top-level \`await\` and \`return\` both work). At run time exactly two of the names declared below are bound: \`tools\` and \`ToolCallError\`. Everything else is a STATIC STUB describing shapes — in particular the \`TypedDict\` classes do NOT exist at run time, so build arguments as plain \`dict\`/\`list\` JSON values: \`await tools.name({"field": 1})\`, never \`FooArgs(field=1)\`, which raises \`NameError\`. Inside the program:
|
||||
|
||||
- Call tools as \`await tools.name(args)\` — subscript access for exotic, reserved, or underscore-leading names: \`await tools["my-tool"](args)\`. Every call resolves to the tool's typed canonical JSON value (each method's return type below). Tool arguments must be lossless JSON.
|
||||
- A FAILED tool call raises \`ToolCallError\`, whose \`toolName\` identifies the failed tool and whose message is human-readable — wrap in \`try/except\` to handle and continue.
|
||||
- Independent read-only calls MAY overlap under \`asyncio.gather\` (safe calls run concurrently; mutating calls run alone, in submission order). Sequence dependent work with \`await\`.
|
||||
- Emit the run's answer with \`print(...)\` and/or a top-level \`return <value>\`; the returned value must be lossless JSON. ONLY what you print and the returned value come back — intermediate tool results never enter the conversation, so extract just what you need.
|
||||
|
||||
The available tools:`
|
||||
|
||||
/**
|
||||
* Render the full `tools:sdk` prompt section under `runtime.language ===
|
||||
* 'python'`: the Python-flavored usage instructions plus one named `TypedDict`
|
||||
* per tool argument or output object (and per nested object) and one awaitable
|
||||
* method per visible tool on a `Tools` protocol — typed args in, the tool's
|
||||
* canonical output value out — with a `tools: Tools` singleton the model calls
|
||||
* into. The `typing` import line lists exactly the symbols the render used.
|
||||
* Deterministic — tools are emitted in lexicographic name order, and class
|
||||
* declarations precede the protocol in that same order (nested classes before
|
||||
* the parent that references them), so an unchanged tool set produces
|
||||
* byte-identical text across assemblies. The sort is not a total order on
|
||||
* byte-equal names, so two schemas sharing a name would render in argument
|
||||
* order; the caller's visible-capability map is keyed by name, so the input
|
||||
* never carries a duplicate.
|
||||
* @param schemas - the tool schemas plus canonical output schemas to declare
|
||||
* (the caller excludes `run_code` itself).
|
||||
* @returns the complete section text.
|
||||
*/
|
||||
export function renderToolsSdkPy(schemas: ToolSdkSchema[]): string {
|
||||
const sorted = [...schemas].sort((a, b) => a.name < b.name ? -1 : a.name > b.name ? 1 : 0)
|
||||
const state: RenderState = { classes: [], usedClassNames: new Set(), nextClassCounter: new Map(), typing: new Set(['Protocol']) }
|
||||
// ONE ordered member stream, matching the documented lexicographic contract
|
||||
// and the TypeScript flavor (which quotes exotic keys in place rather than
|
||||
// partitioning them out). Interleaving is free here: a comment line between
|
||||
// two `async def` lines is not a statement, so it changes nothing about how
|
||||
// the class body parses.
|
||||
const members: string[] = []
|
||||
let statements = 0
|
||||
for (const schema of sorted) {
|
||||
const argType = renderType(schema.parameters, `${camelCase(schema.name)}Args`, state)
|
||||
const outputType = renderType(schema.output, `${camelCase(schema.name)}Output`, state)
|
||||
if (isBareIdentifier(schema.name) && !RESERVED.has(schema.name) && !schema.name.startsWith('_')) {
|
||||
// A docstring only documents its method when it is the FIRST statement
|
||||
// of that method's body. Emitted before the `async def` it would instead
|
||||
// become the `Tools` class docstring (for the first tool) or a dead
|
||||
// expression (for every later one), leaving every method undocumented —
|
||||
// and under `mode: 'code'` this SDK is the model's only description of
|
||||
// what a tool does. A docstring is a complete body, so the `...` stub is
|
||||
// only for the description-less case.
|
||||
const doc = docLines(schema.description, 2)
|
||||
members.push(doc.length > 0
|
||||
? `${pad(1)}async def ${schema.name}(self, args: ${argType}) -> ${outputType}:`
|
||||
: `${pad(1)}async def ${schema.name}(self, args: ${argType}) -> ${outputType}: ...`)
|
||||
members.push(...doc)
|
||||
statements += 1
|
||||
} else {
|
||||
// Not reachable as ``tools.name`` — the model reaches it via
|
||||
// ``tools[name]``. Exotic names and hard keywords are not legal
|
||||
// attributes at all; an underscore-leading name (``_foo``) IS a legal
|
||||
// attribute and is routed here anyway, because the forms that break
|
||||
// split three ways — a non-dunder ``__token`` name-mangles at the CALL
|
||||
// site, a dunder that exists on ``object``/``type`` (``__class__``,
|
||||
// ``__doc__``) resolves before ``__getattr__`` ever runs, and implicit
|
||||
// special-method lookup skips the hook entirely — and one rule over the
|
||||
// whole family costs nothing while a per-form rule would have to
|
||||
// enumerate them (see {@link RESERVED}). The stub lists it as a subscript comment
|
||||
// (referencing the named TypedDicts too) so a reader sees what is
|
||||
// accessible; runtime resolution goes through the proxy's __getitem__.
|
||||
members.push(`${pad(1)}# tools[${JSON.stringify(schema.name)}](args: ${argType}) -> ${outputType}`)
|
||||
const description = describe(schema)
|
||||
if (description !== undefined) members.push(`${pad(1)}# ${description}`)
|
||||
}
|
||||
}
|
||||
// Subscript entries are COMMENTS, not statements: a class body of only
|
||||
// comments fails to parse, so `pass` is required whenever no method was
|
||||
// emitted — including the subscript-only tool set.
|
||||
const bodyLines = statements > 0 ? members : [`${pad(1)}pass`, ...members]
|
||||
const body = bodyLines.join('\n')
|
||||
const imports = TYPING_ORDER.filter(symbol => state.typing.has(symbol))
|
||||
const classBlock = state.classes.length > 0 ? `${state.classes.join('\n\n')}\n\n` : ''
|
||||
const errorDeclaration = 'class ToolCallError(Exception):\n toolName: str'
|
||||
const declaration = `from typing import ${imports.join(', ')}\n\n${errorDeclaration}\n\n${classBlock}class Tools(Protocol):\n${body}\n\ntools: Tools`
|
||||
return `${SDK_INSTRUCTIONS}\n\n\`\`\`python\n${declaration}\n\`\`\``
|
||||
}
|
||||
@@ -262,7 +262,10 @@ The available tools:`
|
||||
* Render the full `tools:sdk` prompt section: the fixed usage instructions
|
||||
* plus one `declare const tools` interface covering every given tool.
|
||||
* Deterministic — tools are emitted in lexicographic name order, so an
|
||||
* unchanged tool set produces byte-identical text across assemblies.
|
||||
* unchanged tool set produces byte-identical text across assemblies. The sort
|
||||
* is not a total order on byte-equal names, so two schemas sharing a name
|
||||
* would render in argument order; the caller's visible-capability map is keyed
|
||||
* by name, so the input never carries a duplicate.
|
||||
* @param schemas - the tool schemas to declare (the caller excludes
|
||||
* `run_code` itself).
|
||||
* @returns the complete section text.
|
||||
|
||||
@@ -335,9 +335,89 @@ describe('mode-aware wire contribution', () => {
|
||||
await expect(systemPrompt.assemble()).rejects.toThrow(/requires a code runtime/)
|
||||
})
|
||||
|
||||
it("rejects every assembly when the runtime's language is not typescript", async () => {
|
||||
const { systemPrompt } = await setup({ mode: 'code', runtime: { language: 'python' } })
|
||||
await expect(systemPrompt.assemble()).rejects.toThrow(/language is "python"/)
|
||||
it('rejects every assembly when the runtime language has no registered SDK renderer', async () => {
|
||||
const { systemPrompt } = await setup({ mode: 'code', runtime: { language: 'ruby' } })
|
||||
await expect(systemPrompt.assemble()).rejects.toThrow(/no SDK renderer registered for runtime language "ruby"/)
|
||||
})
|
||||
|
||||
it('assembles under a python runtime by picking the Python SDK renderer', async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'code', runtime: { language: 'python' } })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
const sdk = assembly.sections.find(section => section.name === 'tools:sdk')
|
||||
expect(sdk?.text).toContain('class Tools(Protocol):')
|
||||
expect(sdk?.text).toContain('async def echo(self, args:')
|
||||
expect(sdk?.text).toContain('top-level `await`')
|
||||
})
|
||||
|
||||
it("assembles under a python runtime in mode 'both' as well, SDK and schema together", async () => {
|
||||
// `both` reaches the same wireSchemas/requireCodeRuntime/SDK-section code
|
||||
// as `code`, so this pins the mode-by-language matrix rather than a
|
||||
// separate path — including that the `wireSchemas` projection behind
|
||||
// `assembly.tools` picks the Python flavor under `both` instead of hitting
|
||||
// the flavor-table guard.
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'both', runtime: { language: 'python' } })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
expect(assembly.sections.find(section => section.name === 'tools:sdk')?.text).toContain('class Tools(Protocol):')
|
||||
const runCodeSchema = assembly.tools.find(tool => tool.name === RUN_CODE_NAME)
|
||||
expect(runCodeSchema?.description).toContain('Execute a Python program')
|
||||
// `both` keeps the native tools alongside run_code; `code` does not.
|
||||
expect(assembly.tools.map(tool => tool.name)).toContain('echo')
|
||||
})
|
||||
|
||||
it('emits a TypeScript-flavored run_code schema under a typescript runtime', async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'code', runtime: { language: 'typescript' } })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
const runCodeSchema = assembly.tools.find(tool => tool.name === RUN_CODE_NAME)
|
||||
expect(runCodeSchema?.description).toContain('Execute a TypeScript program')
|
||||
expect(runCodeSchema?.description).toContain('BODY of an')
|
||||
const codeParam = (runCodeSchema?.parameters as { properties: { code: { description: string } } }).properties.code
|
||||
expect(codeParam.description).toBe('The program: the body of an async TypeScript function.')
|
||||
})
|
||||
|
||||
it('emits a Python-flavored run_code schema under a python runtime (matches the SDK language)', async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'code', runtime: { language: 'python' } })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
const runCodeSchema = assembly.tools.find(tool => tool.name === RUN_CODE_NAME)
|
||||
expect(runCodeSchema?.description).toContain('Execute a Python program')
|
||||
expect(runCodeSchema?.description).toContain('`return <value>`')
|
||||
expect(runCodeSchema?.description).not.toContain('TypeScript')
|
||||
const codeParam = (runCodeSchema?.parameters as { properties: { code: { description: string } } }).properties.code
|
||||
expect(codeParam.description).toBe('The program: the body of an async Python function.')
|
||||
})
|
||||
|
||||
it('resolves the run_code schema flavor lazily and fails loud on a language absent from the flavor table', async () => {
|
||||
// The flavor getter reads the runtime directly (peekRuntime), so it — not
|
||||
// requireCodeRuntime — owns the flavor-table guard. Keeping
|
||||
// RUN_CODE_FLAVORS in step with SDK_RENDERERS is the compiler's job (both
|
||||
// are `satisfies`-checked against CodeSdkLanguage), so what the guard
|
||||
// covers is a mounted runtime naming a language absent from both tables,
|
||||
// which throws when the schema is projected. Assembly's
|
||||
// requireCodeRuntime rejects such a language earlier; this reaches the
|
||||
// guard on its own.
|
||||
const { ctx } = await setup({ mode: 'code', runtime: { language: 'ruby' } })
|
||||
const definition = ctx.tools.get(RUN_CODE_NAME)
|
||||
// Names the known languages, symmetric with the SDK_RENDERERS guard: this
|
||||
// is the reachable rejection, so it must be at least as diagnosable.
|
||||
expect(() => definition?.description)
|
||||
.toThrow(/no run_code schema flavor registered for runtime language "ruby" \(known: "typescript", "python"\)/)
|
||||
})
|
||||
|
||||
it('degrades the run_code flavor to TypeScript when no runtime is mounted', async () => {
|
||||
// Any reader of the definition without a mounted runtime lands here; the
|
||||
// shipped one is the tool-catalog generator, which boots the registry under
|
||||
// `mode: code` and reads run_code's schema WITHOUT a runtime. peekRuntime
|
||||
// returns undefined there, so the flavor getter degrades to the TS default
|
||||
// rather than throwing. None of those readers feeds a model: assembly goes
|
||||
// through wireSchemas, which requires a runtime first.
|
||||
const { ctx } = await setup({ mode: 'code', runtime: false })
|
||||
const definition = ctx.tools.get(RUN_CODE_NAME)
|
||||
expect(definition?.description).toContain('Execute a TypeScript program')
|
||||
const params = definition?.parameters as { properties: { code: { description: string } } }
|
||||
expect(params.properties.code.description).toBe('The program: the body of an async TypeScript function.')
|
||||
})
|
||||
|
||||
it("rejects the assembly when toolOrder names a native tool that mode 'code' no longer contributes", async () => {
|
||||
|
||||
1163
packages/core/tools/tests/py-types.spec.ts
Normal file
1163
packages/core/tools/tests/py-types.spec.ts
Normal file
File diff suppressed because it is too large
Load Diff
@@ -45,6 +45,7 @@
|
||||
"@deepseek-ai/dsh-session-persistence-sqlite": "workspace:^",
|
||||
"@deepseek-ai/dsh-subprocess": "workspace:^",
|
||||
"@deepseek-ai/dsh-tool-subagent": "workspace:^",
|
||||
"@deepseek-ai/dsh-tool-todo": "workspace:^",
|
||||
"@deepseek-ai/dsh-tool-web": "workspace:^",
|
||||
"cordis": "^4.0.0-rc.7"
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ import type { Config as CodexHooksConfig } from '@deepseek-ai/dsh-hooks-codex'
|
||||
import type { Config as JsonlConfig } from '@deepseek-ai/dsh-session-persistence-jsonl'
|
||||
import type { Config as SqliteConfig } from '@deepseek-ai/dsh-session-persistence-sqlite'
|
||||
import type { Config as ToolSubagentConfig } from '@deepseek-ai/dsh-tool-subagent'
|
||||
import type { Config as ToolTodoConfig } from '@deepseek-ai/dsh-tool-todo'
|
||||
import type { Config as ToolWebConfig } from '@deepseek-ai/dsh-tool-web'
|
||||
import type { ProjectProfile } from '../../project/types.ts'
|
||||
import { defineFeatures } from '../define-feature.ts'
|
||||
@@ -126,7 +127,12 @@ config:
|
||||
id: 'default',
|
||||
label: 'todo_write tool',
|
||||
default: true,
|
||||
resources: [{ kind: 'npm-cordis-config-entry', id: 'tool-todo', package: '@deepseek-ai/dsh-tool-todo' }],
|
||||
resources: [{
|
||||
kind: 'npm-cordis-config-entry',
|
||||
id: 'tool-todo',
|
||||
package: '@deepseek-ai/dsh-tool-todo',
|
||||
config: { allowParallelInProgress: true } satisfies ToolTodoConfig,
|
||||
}],
|
||||
}],
|
||||
},
|
||||
{
|
||||
|
||||
@@ -27,6 +27,9 @@
|
||||
{
|
||||
"path": "../../subagent/tool-subagent"
|
||||
},
|
||||
{
|
||||
"path": "../../todo/tool-todo"
|
||||
},
|
||||
{
|
||||
"path": "../../web/tool-web"
|
||||
},
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/todo/tool-todo/README.md
|
||||
README.md: 456d4a08d88b145d574362ffa0874faef9167b22
|
||||
README.zh.md: ec37682773e50c3f153525f6c2b6b6cce583144f
|
||||
README.md: 914e89a000e4bb87ebd7844f05db3809c6726528
|
||||
README.zh.md: c88dbf976fa5110028fcc964ab9aa8efcc3244d3
|
||||
|
||||
@@ -14,9 +14,15 @@ Registers one tool, `todo_write(todos: [{ content, status }])`, on `ctx.tools`.
|
||||
|
||||
The list belongs to the ONE agent session that called the tool. There is no subagent/shared/swarm scope: a non-agent caller (no `exec.agent`) has nowhere to write the list and is rejected. This is a deliberate scope limit — see the Agent Note.
|
||||
|
||||
## Configuration
|
||||
|
||||
`allowParallelInProgress` is required: every composition must choose whether several todos may be `in_progress` at once. It is a deployment choice, not a fixed rule: whether concurrent active tasks are legitimate depends on runtime concurrency the tool cannot observe. Use `true` for agents that may fan out work and `false` to enforce the single-active discipline.
|
||||
|
||||
The flag moves the model-facing instruction and the accepted input together — `true` asks the model to mark every actively worked task and accepts any number, `false` asks for exactly one and rejects a call marking more with `Error: invalid todos: at most one task may be in_progress (got <n>)`. The durable-log invariant does NOT follow it: a log written while parallel work was allowed must still replay after a deployment tightens the policy, so the invariant stays silent on the active count.
|
||||
|
||||
## Validation
|
||||
|
||||
Beyond the schema's type/required/enum checks, `execute` rejects an empty or duplicate `content`, more than one `in_progress` task (a coherent plan has at most one task active), and any item key beyond `content`/`status` — an extended item shape (ids, nesting) fails loud instead of silently flattening, keeping the logged snapshot equal to what the model believes it wrote. Ordering and the discipline of keeping the list current are left to the model via the tool description.
|
||||
Beyond the schema's type/required/enum checks, `execute` rejects an empty or duplicate `content`, and any item key beyond `content`/`status` — an extended item shape (ids, nesting) fails loud instead of silently flattening, keeping the logged snapshot equal to what the model believes it wrote. How many tasks may be `in_progress` at once is the deployment's call (§ Configuration): a composition that chooses `true` permits parallel work (concurrent subagents, background commands) to mark several tasks simultaneously. Ordering and the discipline of keeping the list current are left to the model via the tool description.
|
||||
|
||||
## Rendering
|
||||
|
||||
@@ -50,7 +56,7 @@ Prefix-stable while the definition and visibility are unchanged. Plugin lifecycl
|
||||
|
||||
#### What the model sees
|
||||
|
||||
Each assistant tool call retains the entire replacement list in its arguments. Success returns exactly `Updated todo list: <pending> pending, <inProgress> in progress, <completed> completed.` Stable failures are ``Error: invalid todo: `content` must be a non-empty string``, `Error: invalid todos: duplicate content "<content>"`, `Error: invalid todos: at most one task may be in_progress, got <count>`, and `Error: todo_write requires an owning agent session`. The full `todo/write` session event is UI and replay state, not a second model message.
|
||||
Each assistant tool call retains the entire replacement list in its arguments. Success returns exactly `Updated todo list: <pending> pending, <inProgress> in progress, <completed> completed.` Stable failures are ``Error: invalid todo: `content` must be a non-empty string``, `Error: invalid todos: duplicate content "<content>"`, `Error: todo_write requires an owning agent session`, and — only where the deployment set `allowParallelInProgress: false` — `Error: invalid todos: at most one task may be in_progress (got <n>)`. The full `todo/write` session event is UI and replay state, not a second model message.
|
||||
|
||||
#### Token effect
|
||||
|
||||
|
||||
@@ -14,9 +14,15 @@
|
||||
|
||||
该列表属于调用工具的唯一 agent 会话。不存在 subagent/共享/swarm scope:非 agent 调用方(没有 `exec.agent`)无处写入列表,因此会被拒绝。这是有意设置的 scope 限制,详见 Agent Note(agent 决策记录)。
|
||||
|
||||
## 配置
|
||||
|
||||
`allowParallelInProgress` 是必填项:每个组合都必须选择是否允许多个 todo 同时处于 `in_progress`。这是部署层的选择而非固定规则:并发的活跃任务是否合理,取决于工具无法观测的运行时并发情况。可能并行展开工作的 agent 使用 `true`,`false` 则强制执行单活跃项纪律。
|
||||
|
||||
该开关会同时改变面向模型的指令与接受的输入——`true` 要求模型标记每个正在推进的任务并接受任意数量;`false` 要求恰好一个,并以 `Error: invalid todos: at most one task may be in_progress (got <n>)` 拒绝标记更多的调用。持久日志不变式**不**跟随它:在允许并行时写下的日志,在部署收紧策略之后仍必须可回放,因此不变式对活跃数量保持沉默。
|
||||
|
||||
## 验证
|
||||
|
||||
除 schema 的类型/必填/枚举检查外,`execute` 还会拒绝空或重复的 `content`、同时存在多个 `in_progress` 任务的情况(连贯计划最多只有一个活跃任务),以及 `content`/`status` 之外的任何条目键——扩展条目形状(id、嵌套)会明确报错而不是被静默压平,保证落日志的快照与模型自认为写入的内容一致。列表的顺序及及时更新由模型依照工具描述负责。
|
||||
除 schema 的类型/必填/枚举检查外,`execute` 还会拒绝空或重复的 `content`,以及 `content`/`status` 之外的任何条目键——扩展条目形状(id、嵌套)会明确报错而不是被静默压平,保证落日志的快照与模型自认为写入的内容一致。同时可以有多少任务处于 `in_progress` 由部署决定(见 § 配置):选择 `true` 的组合允许并行工作(并发 subagent、后台命令)同时将多个任务标记为 `in_progress`。列表的顺序及及时更新由模型依照工具描述负责。
|
||||
|
||||
## 渲染
|
||||
|
||||
@@ -50,7 +56,7 @@
|
||||
|
||||
#### 模型看到的内容
|
||||
|
||||
每个 assistant 工具调用都会在参数中保留整个替换列表。成功时原样返回 `Updated todo list: <pending> pending, <inProgress> in progress, <completed> completed.`。稳定失败文本为 ``Error: invalid todo: `content` must be a non-empty string``、`Error: invalid todos: duplicate content "<content>"`、`Error: invalid todos: at most one task may be in_progress, got <count>` 和 `Error: todo_write requires an owning agent session`。完整 `todo/write` 会话事件是 UI 与回放状态,而非第二条模型消息。
|
||||
每个 assistant 工具调用都会在参数中保留整个替换列表。成功时原样返回 `Updated todo list: <pending> pending, <inProgress> in progress, <completed> completed.`。稳定失败文本为 ``Error: invalid todo: `content` must be a non-empty string``、`Error: invalid todos: duplicate content "<content>"`、`Error: todo_write requires an owning agent session`,以及——仅在部署设置了 `allowParallelInProgress: false` 时——`Error: invalid todos: at most one task may be in_progress (got <n>)`。完整 `todo/write` 会话事件是 UI 与回放状态,而非第二条模型消息。
|
||||
|
||||
#### Token 影响
|
||||
|
||||
|
||||
@@ -30,6 +30,7 @@
|
||||
],
|
||||
"license": "BSD-3-Clause",
|
||||
"dependencies": {
|
||||
"schemastery": "^3.18.0",
|
||||
"zod": "^4.4.3"
|
||||
},
|
||||
"peerDependencies": {
|
||||
@@ -41,6 +42,8 @@
|
||||
"cordis": "^4.0.0-rc.7"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@cordisjs/plugin-include": "workspace:^",
|
||||
"@cordisjs/plugin-loader": "workspace:^",
|
||||
"@deepseek-ai/dsh-agent": "workspace:^",
|
||||
"@deepseek-ai/dsh-agent-loop": "workspace:^",
|
||||
"@deepseek-ai/dsh-agent-loop-testkit": "workspace:^",
|
||||
|
||||
@@ -6,7 +6,8 @@
|
||||
*/
|
||||
|
||||
import type { Context } from 'cordis'
|
||||
import { z } from 'zod'
|
||||
import z from 'schemastery'
|
||||
import { z as zod } from 'zod'
|
||||
import type { ZodType } from 'zod'
|
||||
import { defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import type { TodoItem } from '@deepseek-ai/dsh-session'
|
||||
@@ -24,29 +25,73 @@ export const inject = ['tools']
|
||||
/** The valid {@link TodoItem} statuses, as a runtime set for input narrowing. */
|
||||
const STATUSES = ['pending', 'in_progress', 'completed'] as const
|
||||
|
||||
const DESCRIPTION =
|
||||
/** Model-facing todo tool configuration. */
|
||||
export interface Config {
|
||||
/**
|
||||
* Required deployment choice for whether several todos may be `in_progress` at once. True suits
|
||||
* agents that run work concurrently — subagents, background commands, workflow fan-out — and the
|
||||
* description then instructs the model to mark every actively worked task. False restores the
|
||||
* single-active discipline: the description asks for exactly one, and a call marking more is
|
||||
* rejected.
|
||||
*/
|
||||
allowParallelInProgress: boolean
|
||||
}
|
||||
|
||||
/** Schemastery configuration for the todo tool consumer. */
|
||||
export const Config: z<Config> = z.object({
|
||||
allowParallelInProgress: z.boolean().required(),
|
||||
})
|
||||
|
||||
const DESCRIPTION_HEAD =
|
||||
'Record and update a structured task list for the current work. Send the ENTIRE '
|
||||
+ 'list every call — it REPLACES the previous list (there are no partial updates, '
|
||||
+ 'no per-item edits). Use it to plan multi-step work and show progress: add one '
|
||||
+ 'todo per concrete step before you start. Keep AT MOST ONE todo `in_progress` '
|
||||
+ 'at a time; while work remains, exactly one active task should be '
|
||||
+ '`in_progress`. Mark a todo `completed` the moment it is done (do not batch '
|
||||
+ 'completions), and allow no `in_progress` item only once all work is complete. '
|
||||
+ 'Skip the list for trivial single-step tasks. Statuses: `pending` '
|
||||
+ '(not started), `in_progress` (being worked on now), `completed` (finished).'
|
||||
+ 'todo per concrete step before you start. '
|
||||
|
||||
const DESCRIPTION_PARALLEL =
|
||||
'Mark every todo being actively worked '
|
||||
+ 'on `in_progress` — several at once when work genuinely runs in parallel (e.g. '
|
||||
+ 'concurrent subagents or background commands), one for sequential work; while '
|
||||
+ 'work remains, at least one task should be `in_progress`. '
|
||||
|
||||
const DESCRIPTION_SINGLE =
|
||||
'Keep AT MOST ONE todo `in_progress` at a '
|
||||
+ 'time; while work remains, exactly one active task should be `in_progress`. '
|
||||
|
||||
const DESCRIPTION_TAIL =
|
||||
'Mark a todo '
|
||||
+ '`completed` the moment it is done (do not batch completions), and allow no '
|
||||
+ '`in_progress` item only once all work is complete. Skip the list for trivial '
|
||||
+ 'single-step tasks. Statuses: `pending` (not started), `in_progress` (being '
|
||||
+ 'worked on now), `completed` (finished).'
|
||||
|
||||
/**
|
||||
* The model-facing description for one activation. The active-status clause is the only part that
|
||||
* varies, because it is the only instruction the parallel policy changes.
|
||||
* @param allowParallel - whether several todos may be `in_progress` at once.
|
||||
* @returns the composed tool description.
|
||||
*/
|
||||
function describe(allowParallel: boolean): string {
|
||||
return DESCRIPTION_HEAD
|
||||
+ (allowParallel ? DESCRIPTION_PARALLEL : DESCRIPTION_SINGLE)
|
||||
+ DESCRIPTION_TAIL
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate the value constraints the ParameterSchemaSpec can't express and build the canonical {@link
|
||||
* TodoItem}[]: trimmed non-empty unique content and at most one in-progress item. The registry
|
||||
* has already enforced the status enum and rejected unknown item keys (`additionalProperties:
|
||||
* false` — the logged snapshot must equal what the model believes it wrote, so a nested/extended
|
||||
* item shape fails loud at the schema boundary instead of silently flattening); the cast below
|
||||
* records that guarantee.
|
||||
* TodoItem}[]: trimmed non-empty unique content, and at most one `in_progress` item unless the
|
||||
* deployment allows parallel work. The registry has already enforced the status enum and rejected
|
||||
* unknown item keys (`additionalProperties: false` — the logged snapshot must equal what the model
|
||||
* believes it wrote, so a nested/extended item shape fails loud at the schema boundary instead of
|
||||
* silently flattening); the cast below records that guarantee.
|
||||
* @param raw - the model-supplied list, already schema-checked.
|
||||
* @param allowParallel - whether several items may be `in_progress` at once.
|
||||
* @returns the canonical list.
|
||||
*/
|
||||
function toTodoList(raw: { content: string; status: string }[]): TodoItem[] {
|
||||
function toTodoList(raw: { content: string; status: string }[], allowParallel: boolean): TodoItem[] {
|
||||
const todos: TodoItem[] = []
|
||||
const seen = new Set<string>()
|
||||
let inProgress = 0
|
||||
let active = 0
|
||||
for (const item of raw) {
|
||||
const content = item.content.trim()
|
||||
if (content.length === 0) {
|
||||
@@ -56,27 +101,32 @@ function toTodoList(raw: { content: string; status: string }[]): TodoItem[] {
|
||||
throw new Error(`invalid todos: duplicate content ${JSON.stringify(content)}`)
|
||||
}
|
||||
seen.add(content)
|
||||
const status = item.status as TodoItem['status']
|
||||
if (status === 'in_progress') inProgress++
|
||||
todos.push({ content, status })
|
||||
if (item.status === 'in_progress') active++
|
||||
todos.push({ content, status: item.status as TodoItem['status'] })
|
||||
}
|
||||
if (inProgress > 1) {
|
||||
throw new Error(`invalid todos: at most one task may be in_progress, got ${inProgress}`)
|
||||
if (!allowParallel && active > 1) {
|
||||
throw new Error(`invalid todos: at most one task may be in_progress (got ${active})`)
|
||||
}
|
||||
return todos
|
||||
}
|
||||
|
||||
/** Wire payload schema of the `todos` projection (whole list or pre-first-write null). */
|
||||
const todosProjectionSchema: ZodType<TodoItem[] | null> = z.union([
|
||||
z.array(z.object({
|
||||
content: z.string(),
|
||||
status: z.union([z.literal('pending'), z.literal('in_progress'), z.literal('completed')]),
|
||||
const todosProjectionSchema: ZodType<TodoItem[] | null> = zod.union([
|
||||
zod.array(zod.object({
|
||||
content: zod.string(),
|
||||
status: zod.union([zod.literal('pending'), zod.literal('in_progress'), zod.literal('completed')]),
|
||||
})),
|
||||
z.null(),
|
||||
zod.null(),
|
||||
])
|
||||
|
||||
/** Register the `todo_write` tool on `ctx.tools` and, when the session-projection seam is composed, the `todos` unit. */
|
||||
export function apply(ctx: Context): void {
|
||||
/**
|
||||
* Register the `todo_write` tool on `ctx.tools` and, when the session-projection seam is composed,
|
||||
* the `todos` unit.
|
||||
* @param ctx - registrant context carrying the tool registry.
|
||||
* @param config - deployment's explicit todo policy.
|
||||
*/
|
||||
export function apply(ctx: Context, config: Config): void {
|
||||
const allowParallel = config.allowParallelInProgress
|
||||
// The unit child activates only when a projection registry is composed
|
||||
// (headless assemblies without the seam stay unaffected). Standing-plan fold:
|
||||
// latest whole todo/write list, cleared by the next turn/start (turn/end keeps
|
||||
@@ -99,7 +149,7 @@ export function apply(ctx: Context): void {
|
||||
})
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'todo_write',
|
||||
description: DESCRIPTION,
|
||||
description: describe(allowParallel),
|
||||
parameters: {
|
||||
todos: {
|
||||
type: 'array',
|
||||
@@ -155,7 +205,7 @@ export function apply(ctx: Context): void {
|
||||
}],
|
||||
},
|
||||
execute(args, exec) {
|
||||
const todos = toTodoList(args.todos)
|
||||
const todos = toTodoList(args.todos, allowParallel)
|
||||
if (!exec.agent) {
|
||||
// The list is per-agent-session state; a non-agent caller (no owning
|
||||
// session) has nowhere to write it. Reject rather than silently no-op.
|
||||
|
||||
@@ -12,11 +12,18 @@ export const name = 'tool-todo-invariant'
|
||||
/** Service required before the companion can reserve package ownership. */
|
||||
export const inject = ['invariants']
|
||||
|
||||
/** Validate one whole-list todo snapshot before it reaches the durable log. */
|
||||
/**
|
||||
* Validate one whole-list todo snapshot before it reaches the durable log.
|
||||
*
|
||||
* Deliberately silent on how many items are `in_progress`. That is the tool's
|
||||
* per-deployment policy (`Config.allowParallelInProgress`), not a durable-shape
|
||||
* rule: a log written while parallel work was allowed must still replay after a
|
||||
* deployment tightens the policy, so tying the invariant to the current config
|
||||
* would reject history that was valid when it was written.
|
||||
*/
|
||||
function validateTodos(value: unknown, fail: InvariantFailure): void {
|
||||
if (!Array.isArray(value)) fail('todo/write todos must be an array')
|
||||
const seen = new Set<string>()
|
||||
let active = 0
|
||||
for (const item of value) {
|
||||
if (typeof item !== 'object' || item === null) fail('todo/write entries must be objects')
|
||||
const { content, status } = item as Record<string, unknown>
|
||||
@@ -28,9 +35,7 @@ function validateTodos(value: unknown, fail: InvariantFailure): void {
|
||||
if (typeof status !== 'string' || !TODO_STATUSES.has(status)) {
|
||||
fail(`todo/write carries unknown status ${JSON.stringify(status)}`)
|
||||
}
|
||||
if (status === 'in_progress') active += 1
|
||||
}
|
||||
if (active > 1) fail(`todo/write contains ${active} in-progress entries; at most one is allowed`)
|
||||
}
|
||||
|
||||
/* jscpd:ignore-start -- package companions share replay and dispatch plumbing */
|
||||
|
||||
@@ -18,7 +18,7 @@ async function harness(adapter: MockAdapter): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await mountAgentLoopTestDependencies(ctx)
|
||||
await ctx.plugin(AgentLoop, { agents: [] })
|
||||
await ctx.plugin(ToolTodo)
|
||||
await ctx.plugin(ToolTodo, { allowParallelInProgress: true })
|
||||
ctx.llm.registerAdapter(['mock'], adapter)
|
||||
return ctx
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import SessionStore, { type Session, type SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import ToolRegistry from '@deepseek-ai/dsh-tools'
|
||||
import * as ToolTodo from '@deepseek-ai/dsh-tool-todo'
|
||||
import * as TodoInvariant from '@deepseek-ai/dsh-tool-todo/invariant'
|
||||
import InvariantService from '@deepseek-ai/dsh-invariants'
|
||||
|
||||
@@ -17,13 +19,22 @@ function event(todos: unknown): SessionEvent {
|
||||
}
|
||||
|
||||
describe('todo snapshot invariants', () => {
|
||||
it('accepts a unique whole-list snapshot with one active item', async () => {
|
||||
const ctx = await setup()
|
||||
expect(() => { ctx.emit('session/event', {} as Session, event([
|
||||
it('accepts historical and live parallel snapshots under the single-active tool policy', async () => {
|
||||
const todos = [
|
||||
{ content: 'Inspect state', status: 'completed' },
|
||||
{ content: 'Apply fix', status: 'in_progress' },
|
||||
{ content: 'Watch background build', status: 'in_progress' },
|
||||
{ content: 'Run checks', status: 'pending' },
|
||||
])) }).not.toThrow()
|
||||
] as const
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await ctx.plugin(ToolTodo, { allowParallelInProgress: false })
|
||||
ctx.sessions.create().append('todo/write', { todos: [...todos] })
|
||||
await ctx.plugin(InvariantService, { enabled: true })
|
||||
|
||||
await expect(ctx.plugin(TodoInvariant).then(() => undefined)).resolves.toBeUndefined()
|
||||
expect(() => { ctx.emit('session/event', {} as Session, event(todos)) }).not.toThrow()
|
||||
})
|
||||
|
||||
it.each([
|
||||
@@ -36,7 +47,6 @@ describe('todo snapshot invariants', () => {
|
||||
[[{ content: 'same', status: 'pending' }, { content: 'same', status: 'completed' }], /repeats content/],
|
||||
[[{ content: 'task', status: 42 }], /unknown status/],
|
||||
[[{ content: 'task', status: 'paused' }], /unknown status/],
|
||||
[[{ content: 'one', status: 'in_progress' }, { content: 'two', status: 'in_progress' }], /at most one/],
|
||||
])('rejects an incoherent durable todo snapshot', async (todos, message) => {
|
||||
const ctx = await setup()
|
||||
expect(() => { ctx.emit('session/event', {} as Session, event(todos)) }).toThrow(message)
|
||||
|
||||
139
packages/todo/tool-todo/tests/loader-composition.spec.ts
Normal file
139
packages/todo/tool-todo/tests/loader-composition.spec.ts
Normal file
@@ -0,0 +1,139 @@
|
||||
// Proves `allowParallelInProgress` is real configurability and not a constant:
|
||||
// the flag is set in a cordis.yml booted through the real Loader, and both faces
|
||||
// it controls — the model-facing description and the accepted input — follow it.
|
||||
import { mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { pathToFileURL } from 'node:url'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import Loader from '@cordisjs/plugin-loader'
|
||||
import Include from '@cordisjs/plugin-include'
|
||||
import { CallId } from '@deepseek-ai/dsh-llm'
|
||||
import { Session, SessionId } from '@deepseek-ai/dsh-session'
|
||||
import AgentRegistry, { Inbox } from '@deepseek-ai/dsh-agent'
|
||||
import type { Agent } from '@deepseek-ai/dsh-agent'
|
||||
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
|
||||
import ToolRegistry from '@deepseek-ai/dsh-tools'
|
||||
import * as ToolTodo from '@deepseek-ai/dsh-tool-todo'
|
||||
|
||||
let root: string | undefined
|
||||
let context: Context | undefined
|
||||
|
||||
afterEach(async () => {
|
||||
await context?.fiber.dispose()
|
||||
context = undefined
|
||||
if (root !== undefined) await rm(root, { recursive: true, force: true })
|
||||
root = undefined
|
||||
})
|
||||
|
||||
function agent(ctx: Context): Agent {
|
||||
const scope = ctx.plugin(() => {})
|
||||
const id = SessionId('todo-loader-agent')
|
||||
const session = Session.create(id)
|
||||
const value: Agent = {
|
||||
id, options: {}, session, inbox: new Inbox(session, { inserted: () => {}, discarded: () => {}, claimed: () => {} }),
|
||||
status: 'idle', ctx: scope.ctx,
|
||||
followup: () => {}, steer: () => {}, inject: () => {}, send: () => {}, cancel() {},
|
||||
runMaintenance: task => task(new AbortController().signal),
|
||||
whenIdle: () => Promise.resolve(),
|
||||
}
|
||||
ctx.agents.register(value)
|
||||
return value
|
||||
}
|
||||
|
||||
function resultText(result: { content: { type: string; text?: string }[] }): string {
|
||||
return result.content.filter(block => block.type === 'text').map(block => block.text).join('')
|
||||
}
|
||||
|
||||
/**
|
||||
* Boot a cordis.yml carrying the given tool-todo config block.
|
||||
* @param configLines - YAML lines nested under the tool's `config:` key.
|
||||
* @returns the booted context.
|
||||
*/
|
||||
async function boot(configLines: readonly string[]): Promise<Context> {
|
||||
root = await mkdtemp(join(tmpdir(), 'dsh-todo-loader-'))
|
||||
const configPath = join(root, 'cordis.yml')
|
||||
await writeFile(configPath, [
|
||||
"- name: '@deepseek-ai/dsh-agent'",
|
||||
"- name: '@deepseek-ai/dsh-system-prompt'",
|
||||
"- name: '@deepseek-ai/dsh-tools'",
|
||||
"- name: '@deepseek-ai/dsh-tool-todo'",
|
||||
...configLines.length > 0 ? [' config:', ...configLines] : [],
|
||||
'',
|
||||
].join('\n'))
|
||||
|
||||
const ctx = new Context()
|
||||
context = ctx
|
||||
ctx.baseUrl = pathToFileURL(root).href + '/'
|
||||
await ctx.plugin(Loader)
|
||||
ctx.loader.builtins.include = Include
|
||||
const modules = new Map<string, unknown>([
|
||||
['@deepseek-ai/dsh-agent', AgentRegistry],
|
||||
['@deepseek-ai/dsh-system-prompt', SystemPrompt],
|
||||
['@deepseek-ai/dsh-tools', ToolRegistry],
|
||||
['@deepseek-ai/dsh-tool-todo', ToolTodo],
|
||||
])
|
||||
ctx.loader.internal = {
|
||||
version: 'v2',
|
||||
async import(specifier: string) {
|
||||
if (!modules.has(specifier)) throw new Error(`unexpected Loader import: ${specifier}`)
|
||||
return modules.get(specifier)
|
||||
},
|
||||
} as unknown as NonNullable<typeof ctx.loader.internal>
|
||||
await ctx.loader.create({ name: 'cordis:include', config: { path: pathToFileURL(configPath).href } })
|
||||
await ctx.loader.await()
|
||||
return ctx
|
||||
}
|
||||
|
||||
const PARALLEL_TODOS = [
|
||||
{ content: 'run subagent a', status: 'in_progress' },
|
||||
{ content: 'run subagent b', status: 'in_progress' },
|
||||
]
|
||||
|
||||
describe('tool-todo real Loader composition through cordis.yml', () => {
|
||||
it('allowParallelInProgress: false narrows the description and rejects a parallel write', async () => {
|
||||
const ctx = await boot([' allowParallelInProgress: false'])
|
||||
const description = ctx.tools.schemas().find(s => s.name === 'todo_write')?.description ?? ''
|
||||
expect(description).toContain('Keep AT MOST ONE todo `in_progress`')
|
||||
expect(description).not.toContain('several at once')
|
||||
|
||||
const owner = agent(ctx)
|
||||
const result = await ctx.tools.execute({
|
||||
signal: new AbortController().signal,
|
||||
callId: CallId('parallel'),
|
||||
name: 'todo_write',
|
||||
arguments: { todos: PARALLEL_TODOS },
|
||||
agent: owner,
|
||||
})
|
||||
expect(result.isError).toBe(true)
|
||||
expect(resultText(result)).toContain('at most one task may be in_progress')
|
||||
expect(owner.session.events.some(e => e.type === 'todo/write')).toBe(false)
|
||||
}, 30_000)
|
||||
|
||||
it('allowParallelInProgress: true permits a parallel write end to end', async () => {
|
||||
const ctx = await boot([' allowParallelInProgress: true'])
|
||||
const description = ctx.tools.schemas().find(s => s.name === 'todo_write')?.description ?? ''
|
||||
expect(description).toContain('several at once when work genuinely runs in parallel')
|
||||
|
||||
const owner = agent(ctx)
|
||||
const result = await ctx.tools.execute({
|
||||
signal: new AbortController().signal,
|
||||
callId: CallId('parallel-enabled'),
|
||||
name: 'todo_write',
|
||||
arguments: { todos: PARALLEL_TODOS },
|
||||
agent: owner,
|
||||
})
|
||||
expect(result.isError).toBe(false)
|
||||
expect(owner.session.events.findLast(e => e.type === 'todo/write')?.data.todos).toEqual(PARALLEL_TODOS)
|
||||
}, 30_000)
|
||||
|
||||
it.each([
|
||||
{ label: 'is omitted', configLines: [], failure: '$.allowParallelInProgress missing required value' },
|
||||
{ label: 'is not boolean', configLines: [' allowParallelInProgress: "no"'], failure: '$.allowParallelInProgress expected boolean' },
|
||||
])('fails loading when allowParallelInProgress $label', async ({ configLines, failure }) => {
|
||||
// The policy is self-contained, so misconfiguration fails at load: the
|
||||
// entry's apply rejects and boot never reaches a running tool.
|
||||
await expect(boot(configLines)).rejects.toThrow(failure)
|
||||
}, 30_000)
|
||||
})
|
||||
@@ -42,7 +42,7 @@ async function harness(withTodoTool: boolean): Promise<Bench> {
|
||||
await ctx.plugin(UserInteractionService)
|
||||
await ctx.plugin(AgentRegistry)
|
||||
await ctx.plugin(SessionProjectionRegistry)
|
||||
if (withTodoTool) await ctx.plugin(ToolTodo)
|
||||
if (withTodoTool) await ctx.plugin(ToolTodo, { allowParallelInProgress: true })
|
||||
const session = ctx.sessions.create()
|
||||
ctx.agents.register({ id: session.id, session, status: 'idle', ctx } as Agent)
|
||||
const api = createApiProxy(ctx, { provider: 'p', model: 'm', cwd: '/tmp', workspaceRoot: '/tmp' })
|
||||
@@ -116,7 +116,7 @@ describe('todos projection provider', () => {
|
||||
it('drops the key when the tool-todo fiber unloads (HMR safety)', async () => {
|
||||
const bench = await harness(false)
|
||||
seedMessage(bench.session)
|
||||
const fiber = await bench.ctx.plugin(ToolTodo)
|
||||
const fiber = await bench.ctx.plugin(ToolTodo, { allowParallelInProgress: true })
|
||||
expect((await bench.tailProjections())?.values).toEqual({ todos: null })
|
||||
await fiber.dispose()
|
||||
expect('todos' in ((await bench.tailProjections())?.values ?? {})).toBe(false)
|
||||
|
||||
@@ -26,11 +26,11 @@ function agentWithSession(id = 'parent-1'): Agent & { session: Session } {
|
||||
return { id: SessionId(id), session } as unknown as Agent & { session: Session }
|
||||
}
|
||||
|
||||
async function setup(): Promise<Context> {
|
||||
async function setup(allowParallelInProgress: boolean): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
await ctx.plugin(tool)
|
||||
await ctx.plugin(tool, { allowParallelInProgress })
|
||||
return ctx
|
||||
}
|
||||
|
||||
@@ -52,7 +52,7 @@ function text(result: { content: { type: string; text?: string }[] }): string {
|
||||
|
||||
describe('dsh-tool-todo', () => {
|
||||
it('registers a `todo_write` tool whose schema is an array of {content,status}', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const schema = ctx.tools.schemas().find(s => s.name === 'todo_write')
|
||||
expect(schema).toBeDefined()
|
||||
const props = (schema!.parameters as { properties?: Record<string, unknown> }).properties ?? {}
|
||||
@@ -65,7 +65,7 @@ describe('dsh-tool-todo', () => {
|
||||
})
|
||||
|
||||
it('appends a todo/write event carrying the whole list to the calling session', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const agent = agentWithSession('writer')
|
||||
const todos: TodoItem[] = [
|
||||
{ content: 'plan', status: 'in_progress' },
|
||||
@@ -85,7 +85,7 @@ describe('dsh-tool-todo', () => {
|
||||
})
|
||||
|
||||
it('stores the trimmed content (the dedupe/length key), not the raw input', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const agent = agentWithSession('trim')
|
||||
const result = await callTodo(ctx, { todos: [{ content: ' plan the work ', status: 'pending' }] }, { agent })
|
||||
expect(result.isError).toBe(false)
|
||||
@@ -95,7 +95,7 @@ describe('dsh-tool-todo', () => {
|
||||
})
|
||||
|
||||
it('replaces the list on a second call (last-write-wins on the log)', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const agent = agentWithSession('writer-2')
|
||||
await callTodo(ctx, { todos: [{ content: 'a', status: 'pending' }] }, { agent })
|
||||
await callTodo(ctx, { todos: [
|
||||
@@ -111,38 +111,99 @@ describe('dsh-tool-todo', () => {
|
||||
})
|
||||
|
||||
it('rejects a malformed status before execute runs (registry arg-validation)', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const result = await callTodo(ctx, { todos: [{ content: 'x', status: 'doing' }] })
|
||||
expect(result.isError).toBe(true)
|
||||
})
|
||||
|
||||
it('rejects a non-array todos argument', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const result = await callTodo(ctx, { todos: 'nope' })
|
||||
expect(result.isError).toBe(true)
|
||||
})
|
||||
|
||||
it('accepts several in_progress items at once (parallel work)', async () => {
|
||||
const ctx = await setup(true)
|
||||
const agent = agentWithSession('parallel')
|
||||
const todos: TodoItem[] = [
|
||||
{ content: 'run subagent a', status: 'in_progress' },
|
||||
{ content: 'run subagent b', status: 'in_progress' },
|
||||
{ content: 'merge results', status: 'pending' },
|
||||
]
|
||||
const result = await callTodo(ctx, { todos }, { agent })
|
||||
expect(result.isError).toBe(false)
|
||||
if (result.isError) throw new Error('expected todo_write success')
|
||||
expect(result.value).toEqual({
|
||||
todos,
|
||||
counts: { pending: 1, inProgress: 2, completed: 0 },
|
||||
})
|
||||
expect(agent.session.events.findLast(e => e.type === 'todo/write')!.data.todos).toEqual(todos)
|
||||
})
|
||||
|
||||
describe('allowParallelInProgress', () => {
|
||||
const parallel = [
|
||||
{ content: 'run subagent a', status: 'in_progress' },
|
||||
{ content: 'run subagent b', status: 'in_progress' },
|
||||
]
|
||||
|
||||
it('false rejects a call marking several items in_progress', async () => {
|
||||
const ctx = await setup(false)
|
||||
const agent = agentWithSession('single-active')
|
||||
const result = await callTodo(ctx, { todos: parallel }, { agent })
|
||||
expect(result.isError).toBe(true)
|
||||
expect(text(result)).toContain('at most one task may be in_progress')
|
||||
// A rejected call must not reach the durable log.
|
||||
expect(agent.session.events.some(e => e.type === 'todo/write')).toBe(false)
|
||||
})
|
||||
|
||||
it('false still accepts one active item', async () => {
|
||||
const ctx = await setup(false)
|
||||
const todos: TodoItem[] = [
|
||||
{ content: 'run subagent a', status: 'in_progress' },
|
||||
{ content: 'run subagent b', status: 'pending' },
|
||||
]
|
||||
const result = await callTodo(ctx, { todos })
|
||||
expect(result.isError).toBe(false)
|
||||
})
|
||||
|
||||
it('true accepts the very list false rejects', async () => {
|
||||
const ctx = await setup(true)
|
||||
const result = await callTodo(ctx, { todos: parallel })
|
||||
expect(result.isError).toBe(false)
|
||||
})
|
||||
|
||||
it('instructs the model to keep at most one active, while true instructs parallel', async () => {
|
||||
const single = await setup(false)
|
||||
const singleDesc = single.tools.schemas().find(s => s.name === 'todo_write')!.description
|
||||
expect(singleDesc).toContain('Keep AT MOST ONE todo `in_progress`')
|
||||
expect(singleDesc).not.toContain('several at once')
|
||||
|
||||
const parallelDesc = (await setup(true)).tools.schemas().find(s => s.name === 'todo_write')!.description
|
||||
expect(parallelDesc).toContain('several at once when work genuinely runs in parallel')
|
||||
expect(parallelDesc).not.toContain('AT MOST ONE')
|
||||
})
|
||||
})
|
||||
|
||||
it.each([
|
||||
{ label: 'empty content', todos: [{ content: ' ', status: 'pending' }], fragment: 'non-empty' },
|
||||
{ label: 'duplicate content', todos: [{ content: 'dup', status: 'pending' }, { content: 'dup', status: 'completed' }], fragment: 'duplicate' },
|
||||
{ label: 'two in_progress', todos: [{ content: 'a', status: 'in_progress' }, { content: 'b', status: 'in_progress' }], fragment: 'in_progress' },
|
||||
{ label: 'unknown item keys', todos: [{ content: 'a', status: 'pending', children: [] }], fragment: 'not a declared property' },
|
||||
])('rejects $label as an isError result', async ({ todos, fragment }) => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const result = await callTodo(ctx, { todos })
|
||||
expect(result.isError).toBe(true)
|
||||
expect(text(result)).toContain(fragment)
|
||||
})
|
||||
|
||||
it('rejects a non-agent caller (the list has no owning session)', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const result = await callTodo(ctx, { todos: [{ content: 'a', status: 'pending' }] }, { agent: undefined })
|
||||
expect(result.isError).toBe(true)
|
||||
expect(text(result)).toContain('owning agent session')
|
||||
})
|
||||
|
||||
it('presents the call with a stable title and the list as raw input', async () => {
|
||||
const ctx = await setup()
|
||||
const ctx = await setup(true)
|
||||
const def = ctx.tools.get('todo_write')!
|
||||
const todos = [{ content: 'a', status: 'pending' }]
|
||||
expect(def.presentCall?.({ todos })).toEqual({ card: 'generic', title: 'Update todo list', kind: 'other', rawInput: todos })
|
||||
@@ -152,7 +213,7 @@ describe('dsh-tool-todo', () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
const fiber = await ctx.plugin(tool)
|
||||
const fiber = await ctx.plugin(tool, { allowParallelInProgress: true })
|
||||
expect(ctx.tools.schemas().some(s => s.name === 'todo_write')).toBe(true)
|
||||
await fiber.dispose()
|
||||
expect(ctx.tools.schemas().some(s => s.name === 'todo_write')).toBe(false)
|
||||
|
||||
Reference in New Issue
Block a user