Merge remote-tracking branch 'origin/master' into xtr/react-loop-simplification

# Conflicts:
#	.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.i18n.yaml
#	.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.i18n.yaml
#	.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.md
#	.agents/notes/implemented/feature/2026-06-18-compaction-capability-seam.zh.md
#	.agents/notes/implemented/feature/2026-07-06-sandbox.i18n.yaml
#	.agents/notes/implemented/feature/2026-07-06-sandbox.md
#	.agents/notes/implemented/feature/2026-07-06-sandbox.zh.md
#	.agents/notes/implemented/feature/2026-07-27-tmux-location-context.i18n.yaml
#	.agents/notes/implemented/simplification/2026-06-20-public-agent-stop-surface.i18n.yaml
#	.agents/notes/implemented/simplification/2026-07-28-remove-synthetic-log-only-turns.i18n.yaml
#	.agents/notes/implemented/simplification/2026-07-30-private-agent-send.i18n.yaml
#	docs/architecture.i18n.yaml
#	docs/architecture.md
#	docs/architecture.zh.md
#	docs/config-catalog.md
#	docs/cordis-catalog/events.md
#	docs/cordis-catalog/services.md
#	docs/core-data-structures/compaction.i18n.yaml
#	docs/core-data-structures/core.i18n.yaml
#	docs/core-data-structures/core.md
#	docs/core-data-structures/core.zh.md
#	docs/core-data-structures/llm-streaming.i18n.yaml
#	docs/core-data-structures/llm-streaming.md
#	docs/core-data-structures/llm-streaming.zh.md
#	docs/core-data-structures/session.i18n.yaml
#	docs/event-producer-consumer.md
#	docs/module-graph.md
#	docs/persistence-catalog.md
#	examples/acp-agent/tests/goal-snapshots/goal-session/session.expected.jsonl
#	examples/acp-agent/tests/snapshots/advanced-toolchain/session.1.jsonl
#	examples/acp-agent/tests/snapshots/advanced-toolchain/session.2.jsonl
#	examples/acp-agent/tests/snapshots/advanced-toolchain/session.jsonl
#	examples/acp-agent/tests/snapshots/bash-spill/session.jsonl
#	examples/acp-agent/tests/snapshots/bash-tool-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/both-mode-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/cancel-tool-calls/session.jsonl
#	examples/acp-agent/tests/snapshots/cancel/session.jsonl
#	examples/acp-agent/tests/snapshots/code-mode-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/code-mode-workspace-context/session.jsonl
#	examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/session.jsonl
#	examples/acp-agent/tests/snapshots/empty-response-retry/session.jsonl
#	examples/acp-agent/tests/snapshots/error-finish/session.jsonl
#	examples/acp-agent/tests/snapshots/escalation-approved/session.jsonl
#	examples/acp-agent/tests/snapshots/escalation-rejected/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-edit/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-escalation-approved/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-glob-sampling/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-policy-reject/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-read-window/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-read/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-write-overwrite/session.jsonl
#	examples/acp-agent/tests/snapshots/fs-write/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-invalid-matcher/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-posttool-block/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-posttool-context/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-pretool-ask/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-pretool-deny/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-promptsubmit-context/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-cc-stop-continue/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-codex-invalid-matcher/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-codex-posttool-block/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-codex-posttool-context/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-codex-pretool-block/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-codex-promptsubmit-context/session.jsonl
#	examples/acp-agent/tests/snapshots/hook-codex-stop-continue/session.jsonl
#	examples/acp-agent/tests/snapshots/lsp-definition/session.jsonl
#	examples/acp-agent/tests/snapshots/multi-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/packed-chunks/session.jsonl
#	examples/acp-agent/tests/snapshots/parallel-tool-calls/session.jsonl
#	examples/acp-agent/tests/snapshots/pty-tools/session.jsonl
#	examples/acp-agent/tests/snapshots/repeat-tool-guard/session.jsonl
#	examples/acp-agent/tests/snapshots/session-query-spill/session.jsonl
#	examples/acp-agent/tests/snapshots/session-sandbox-root/session.jsonl
#	examples/acp-agent/tests/snapshots/session-title-after-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/skill-load/session.jsonl
#	examples/acp-agent/tests/snapshots/subagent-depth-two-rejection/session.1.jsonl
#	examples/acp-agent/tests/snapshots/subagent-depth-two-rejection/session.2.jsonl
#	examples/acp-agent/tests/snapshots/subagent-depth-two-rejection/session.jsonl
#	examples/acp-agent/tests/snapshots/subagent-fork/session.1.jsonl
#	examples/acp-agent/tests/snapshots/subagent-fork/session.jsonl
#	examples/acp-agent/tests/snapshots/subagent-mixed/session.1.jsonl
#	examples/acp-agent/tests/snapshots/subagent-mixed/session.2.jsonl
#	examples/acp-agent/tests/snapshots/subagent-mixed/session.jsonl
#	examples/acp-agent/tests/snapshots/subagent-multi/session.1.jsonl
#	examples/acp-agent/tests/snapshots/subagent-multi/session.2.jsonl
#	examples/acp-agent/tests/snapshots/subagent-multi/session.jsonl
#	examples/acp-agent/tests/snapshots/subagent-spawn/session.1.jsonl
#	examples/acp-agent/tests/snapshots/subagent-spawn/session.jsonl
#	examples/acp-agent/tests/snapshots/text-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/todo-write/session.jsonl
#	examples/acp-agent/tests/snapshots/tool-call-turn/session.jsonl
#	examples/acp-agent/tests/snapshots/web-fetch/session.jsonl
#	examples/acp-agent/tests/snapshots/workflow-run/session.1.jsonl
#	examples/acp-agent/tests/snapshots/workflow-run/session.jsonl
#	examples/acp-agent/tests/snapshots/workspace-context/session.jsonl
#	examples/acp-agent/tests/snapshots/workspace-edit/session.jsonl
#	examples/headless-agent/tests/semantic-checkpoint-snapshots/tool-outcome-unknown/session.expected.jsonl
#	examples/headless-agent/tests/snapshots/advanced-toolchain/session.1.jsonl
#	examples/headless-agent/tests/snapshots/advanced-toolchain/session.2.jsonl
#	examples/headless-agent/tests/snapshots/advanced-toolchain/session.jsonl
#	examples/headless-agent/tests/snapshots/advanced-toolchain/stream-json.expected.jsonl
#	examples/headless-agent/tests/snapshots/goal-tools/stream-json.expected.jsonl
#	examples/headless-agent/tests/snapshots/missing-credential/stream-json.expected.jsonl
#	examples/headless-agent/tests/snapshots/provider-retry/stream-json.expected.jsonl
#	examples/headless-agent/tests/snapshots/pty-tools/session.jsonl
#	examples/headless-agent/tests/snapshots/pty-tools/stream-json.expected.jsonl
#	examples/headless-agent/tests/snapshots/ralph-loop/stream-json.expected.jsonl
#	examples/headless-agent/tests/subagent-inheritance-snapshots/parent-override/child.expected.jsonl
#	examples/headless-agent/tests/subagent-inheritance-snapshots/parent-override/parent.expected.jsonl
#	examples/jsonrpc-agent/tests/snapshots/bash-tool/notifications.expected.jsonl
#	examples/jsonrpc-agent/tests/snapshots/bash-tool/session.jsonl
#	examples/jsonrpc-agent/tests/snapshots/persistent-tools/notifications.expected.jsonl
#	examples/jsonrpc-agent/tests/snapshots/persistent-tools/session.jsonl
#	examples/jsonrpc-agent/tests/snapshots/subagent-spawn/notifications.expected.jsonl
#	examples/jsonrpc-agent/tests/snapshots/subagent-spawn/session.1.jsonl
#	examples/jsonrpc-agent/tests/snapshots/subagent-spawn/session.jsonl
#	examples/jsonrpc-agent/tests/snapshots/text-turn/notifications.expected.jsonl
#	examples/jsonrpc-agent/tests/snapshots/text-turn/session.jsonl
#	packages/client/runtime/README.i18n.yaml
#	packages/client/runtime/src/client/sessions/request-inspection.ts
#	packages/compact/compact-basic/README.i18n.yaml
#	packages/compact/compact-basic/README.md
#	packages/compact/compact-basic/README.zh.md
#	packages/compact/compact-basic/src/index.ts
#	packages/context/time-context/tests/time-context.spec.ts
#	packages/context/tmux-context/README.i18n.yaml
#	packages/context/tmux-context/tests/tmux-context.spec.ts
#	packages/context/workspace-context/tests/workspace-context.spec.ts
#	packages/cordis/tool-cordis/src/api-catalog.ts
#	packages/core/agent-loop/README.i18n.yaml
#	packages/core/agent-loop/README.md
#	packages/core/agent-loop/README.zh.md
#	packages/core/agent-loop/src/agent.ts
#	packages/core/agent/README.i18n.yaml
#	packages/core/agent/README.md
#	packages/core/agent/README.zh.md
#	packages/core/agent/src/types.ts
#	packages/core/session/README.i18n.yaml
#	packages/core/session/README.md
#	packages/core/session/README.zh.md
#	packages/fs/tool-str-replace-editor/tests/tools.spec.ts
#	packages/goal/command-goal/tests/command-goal.spec.ts
#	packages/goal/goal/tests/goal.spec.ts
#	packages/host/apiproxy/README.i18n.yaml
#	packages/host/apiproxy/README.md
#	packages/host/apiproxy/README.zh.md
#	packages/host/apiproxy/src/api/index.ts
#	packages/host/apiproxy/tests/api-proxy-workspace.spec.ts
#	packages/llm/llm/README.i18n.yaml
#	packages/llm/llm/README.md
#	packages/llm/llm/README.zh.md
#	packages/llm/llm/src/index.ts
#	packages/pty/pty-local/tests/index.spec.ts
#	packages/pty/pty-local/tests/local.spec.ts
#	packages/pty/pty/tests/service.spec.ts
#	packages/pty/tool-bash-persistent/tests/loader-composition.spec.ts
#	packages/pty/tool-bash-persistent/tests/tools.spec.ts
#	packages/pty/tool-pty/tests/loader-composition.spec.ts
#	packages/pty/tool-pty/tests/tools.spec.ts
#	packages/session-persistence/session-checkpoint-policy/tests/crash-recovery.e2e.ts
#	packages/skill/tool-skill/tests/tool-skill.spec.ts
#	packages/tasks/tasks-local/tests/tasks.spec.ts
#	packages/ui/tui/README.i18n.yaml
#	packages/ui/tui/tests/tui.spec.ts
#	packages/ui/user-approval/src/index.ts
#	packages/ui/user-approval/tests/approval.spec.ts
This commit is contained in:
_Kerman
2026-07-31 22:16:40 +08:00
1010 changed files with 105572 additions and 5504 deletions

View File

@@ -1,9 +1,9 @@
import { createUserMessage } from '@deepseek-ai/dsh-llm'
/**
* Tests for the queue-aware `Agent.cancel()` primitive. `cancel()` is the broad verb — it
* clears queued + steering work, aborts the active turn, and drops work not yet claimed by the
* driver without leaking cancellation into a replacement prompt. The suite covers every landing
* window plus signal reset and `whenIdle()` quiescence.
* Tests for the queue-aware `Agent.cancel()` primitive. The default clears
* queued and steering work, while `keepInbox` preserves pending input and
* resumes waking turns after the active turn reaches quiescence. The suite
* covers every landing window plus signal reset and `whenIdle()` quiescence.
* @module dsh-agent-loop/tests/cancel
*/
@@ -284,6 +284,41 @@ describe('Agent.cancel()', () => {
expect(adapter.requests).toHaveLength(1)
})
it('cancel({ keepInbox: true }) aborts the active turn and drains the queued tail in FIFO order', async () => {
const adapter = new MockAdapter([
'hang',
textResponse('second reply'),
textResponse('third reply'),
])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('keep-inbox-running'), { provider: 'mock', model: 'mock' })
const reasons: TurnEndReason[] = []
const discards: unknown[] = []
ctx.on('session/event', (session, event) => {
if (session === agent.session && event.type === 'turn/end') reasons.push(event.data.reason)
})
ctx.on('agent/inbox/discard', (subject, items) => {
if (subject === agent) discards.push(items)
})
send(agent, 'active')
await new Promise(resolve => setTimeout(resolve, 30))
send(agent, 'queued second')
send(agent, 'queued third')
const idle = agent.whenIdle()
agent.cancel({ kind: 'user' }, { keepInbox: true })
await idle
expect(discards).toEqual([])
expect(userTexts(agent)).toEqual(['active', 'queued second', 'queued third'])
expect(reasons).toEqual([
{ kind: 'aborted' },
{ kind: 'completed' },
{ kind: 'completed' },
])
expect(adapter.requests).toHaveLength(3)
})
it('cancel from an assistant/message observer skips execution but balances replay', async () => {
const adapter = new MockAdapter([
toolCallResponse('c1', 'danger', {}),

View File

@@ -258,7 +258,7 @@ describe('agent loop', () => {
// NO system field at all (not an empty string).
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
ctx.on('system-prompt/assemble', async () => ({ sections: [], tools: [], variables: {} }))
ctx.on('system-prompt/assemble', async () => ({ sections: [], contexts: [], tools: [], variables: {} }))
const agent = ctx.agentLoop.create(SessionId('a-no-system'), { provider: 'mock', model: 'mock' })
send(agent, 'hi')
@@ -268,6 +268,178 @@ describe('agent loop', () => {
expect('system' in adapter.requests[0]!).toBe(false)
})
it('materializes changed runtime context at the history tail without rewriting the system header', async () => {
const adapter = new MockAdapter([
textResponse('one'),
textResponse('two'),
textResponse('three'),
textResponse('four'),
textResponse('five'),
])
const ctx = await harness(adapter)
let mode = 'read-only'
const dispose = ctx.systemPrompt.context({ name: 'policy', order: 0, text: () => `Mode: ${mode}.` })
const agent = ctx.agentLoop.create(SessionId('a-runtime-context'), { provider: 'mock', model: 'mock' })
const contextEvents = () => agent.session.events.flatMap(event =>
event.type === 'user/message'
&& event.data.source.kind === 'plugin'
&& event.data.source.plugin === '@deepseek-ai/dsh-system-prompt'
? [event]
: [])
send(agent, 'first')
await waitForIdle(ctx, agent)
expect(contextEvents()).toHaveLength(1)
expect(contextEvents()[0]?.data.content).toEqual([{
type: 'text',
text: 'Current runtime context. This snapshot supersedes earlier runtime-context snapshots.\n\nMode: read-only.',
}])
send(agent, 'unchanged')
await waitForIdle(ctx, agent)
expect(contextEvents()).toHaveLength(1)
mode = 'danger-full-access'
send(agent, 'changed')
await waitForIdle(ctx, agent)
expect(contextEvents()).toHaveLength(2)
const changedBlock = contextEvents()[1]?.data.content[0]
expect(changedBlock?.type).toBe('text')
if (changedBlock?.type !== 'text') throw new Error('changed runtime context is not text')
expect(changedBlock.text).toContain('danger-full-access')
dispose()
send(agent, 'cleared')
await waitForIdle(ctx, agent)
expect(contextEvents()).toHaveLength(3)
expect(contextEvents()[2]?.data.content).toEqual([{
type: 'text',
text: 'Current runtime context: none. Earlier runtime-context snapshots no longer apply.',
}])
send(agent, 'still clear')
await waitForIdle(ctx, agent)
expect(contextEvents()).toHaveLength(3)
expect(adapter.requests.map(request => request.system)).toEqual(Array(5).fill(adapter.requests[0]?.system))
expect(agent.session.events.filter(event => event.type === 'request/header')).toHaveLength(1)
})
it('re-emits unchanged runtime context when a surface replacement removed the retained snapshot', async () => {
const adapter = new MockAdapter([textResponse('one'), textResponse('two')])
const ctx = await harness(adapter)
ctx.systemPrompt.context({ name: 'policy', order: 0, text: 'Mode: read-only.' })
const agent = ctx.agentLoop.create(SessionId('a-runtime-context-compacted'), { provider: 'mock', model: 'mock' })
send(agent, 'first')
await waitForIdle(ctx, agent)
const contextEvent = agent.session.events.find(event =>
event.type === 'user/message'
&& event.data.source.kind === 'plugin'
&& event.data.source.plugin === '@deepseek-ai/dsh-system-prompt')
if (contextEvent?.type !== 'user/message') throw new Error('first turn did not materialize runtime context')
agent.session.append('user/message', createUserMessage({
content: [{ type: 'text', text: 'compacted summary' }],
source: { kind: 'plugin', plugin: 'test-compaction' },
}), {
surfaceOp: { op: 'replace', start: contextEvent.seq, end: contextEvent.seq },
sourceEventSeqs: [contextEvent.seq],
})
send(agent, 'after compaction')
await waitForIdle(ctx, agent)
const runtimeContexts = agent.session.events.flatMap(event =>
event.type === 'user/message'
&& event.data.source.kind === 'plugin'
&& event.data.source.plugin === '@deepseek-ai/dsh-system-prompt'
? [event]
: [])
expect(runtimeContexts).toHaveLength(2)
expect(adapter.requests[1]?.messages.some(message =>
message.source.kind === 'plugin'
&& message.source.plugin === '@deepseek-ai/dsh-system-prompt')).toBe(true)
})
it('clears compacted runtime context after the active set becomes empty', async () => {
const adapter = new MockAdapter([textResponse('one'), textResponse('two')])
const ctx = await harness(adapter)
const dispose = ctx.systemPrompt.context({ name: 'policy', order: 0, text: 'Mode: read-only.' })
const agent = ctx.agentLoop.create(SessionId('a-runtime-context-compacted-clear'), { provider: 'mock', model: 'mock' })
send(agent, 'first')
await waitForIdle(ctx, agent)
const contextEvent = agent.session.events.find(event =>
event.type === 'user/message'
&& event.data.source.kind === 'plugin'
&& event.data.source.plugin === '@deepseek-ai/dsh-system-prompt')
if (contextEvent?.type !== 'user/message') throw new Error('first turn did not materialize runtime context')
agent.session.append('user/message', createUserMessage({
content: [{ type: 'text', text: 'summary retaining old mode: read-only' }],
source: { kind: 'plugin', plugin: 'test-compaction' },
}), {
surfaceOp: { op: 'replace', start: contextEvent.seq, end: contextEvent.seq },
sourceEventSeqs: [contextEvent.seq],
})
dispose()
send(agent, 'after compaction')
await waitForIdle(ctx, agent)
const clearing = adapter.requests[1]?.messages.find(message =>
message.source.kind === 'plugin'
&& message.source.plugin === '@deepseek-ai/dsh-system-prompt')
expect(clearing?.content).toEqual([{
type: 'text',
text: 'Current runtime context: none. Earlier runtime-context snapshots no longer apply.',
}])
})
it('does not clear runtime context after an unrelated replacement', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a-runtime-context-unrelated-compaction'), { provider: 'mock', model: 'mock' })
const original = agent.session.append('user/message', createUserMessage({
content: [{ type: 'text', text: 'old context' }],
source: { kind: 'plugin', plugin: 'test-context' },
}), { surfaceOp: 'append' })
agent.session.append('user/message', createUserMessage({
content: [{ type: 'text', text: 'compacted summary' }],
source: { kind: 'plugin', plugin: 'test-compaction' },
}), {
surfaceOp: { op: 'replace', start: original.seq, end: original.seq },
sourceEventSeqs: [original.seq],
})
send(agent, 'after compaction')
await waitForIdle(ctx, agent)
expect(adapter.requests[0]?.messages.some(message =>
message.source.kind === 'plugin'
&& message.source.plugin === '@deepseek-ai/dsh-system-prompt')).toBe(false)
})
it('replaces a malformed retained runtime-context message with the current complete snapshot', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
ctx.systemPrompt.context({ name: 'policy', order: 0, text: 'Mode: read-only.' })
const agent = ctx.agentLoop.create(SessionId('a-runtime-context-malformed'), { provider: 'mock', model: 'mock' })
agent.session.append('user/message', createUserMessage({
content: [{ type: 'text', text: 'broken' }, { type: 'text', text: 'snapshot' }],
source: { kind: 'plugin', plugin: '@deepseek-ai/dsh-system-prompt' },
}), { surfaceOp: 'append' })
send(agent, 'repair context')
await waitForIdle(ctx, agent)
const runtimeContexts = agent.session.events.flatMap(event =>
event.type === 'user/message'
&& event.data.source.kind === 'plugin'
&& event.data.source.plugin === '@deepseek-ai/dsh-system-prompt'
? [event]
: [])
expect(runtimeContexts).toHaveLength(2)
expect(runtimeContexts[1]?.data.content).toEqual([{
type: 'text',
text: 'Current runtime context. This snapshot supersedes earlier runtime-context snapshots.\n\nMode: read-only.',
}])
})
it('records raw chunks for replay as assistant/chunk session events', async () => {
const adapter = new MockAdapter([textResponse('abc')])
const ctx = await harness(adapter)

View File

@@ -8,7 +8,7 @@
import { describe, expect, it } from 'vitest'
import { Context } from 'cordis'
import LlmService, { createUserMessage, LlmError, ReasoningEffortId } from '@deepseek-ai/dsh-llm'
import type { GenerateOptions, LlmModelReasoningInfo, LlmResolvedModelInfo } from '@deepseek-ai/dsh-llm'
import type { GenerateOptions, LlmModelReasoningInfo, LlmResolvedModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm'
import SessionStore, { Session, SessionId, foldRequestHeader } from '@deepseek-ai/dsh-session'
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
import ToolRegistry, { defineContentToolFixture } from '@deepseek-ai/dsh-tools'
@@ -613,3 +613,95 @@ describe('request stability across the loop', () => {
})
})
})
describe('request/context capacity records', () => {
/** Adapter advertising a per-model capacity, keyed by model id. */
function capacityAdapter(windows: Record<string, number>, script: StreamChunk[][]): MockAdapter {
return new class extends MockAdapter {
override resolveModel(provider: string, model: string): Promise<LlmResolvedModelInfo> {
const contextWindow = windows[model]
return Promise.resolve({
provider,
id: model,
name: model,
...contextWindow === undefined ? {} : { context: { contextWindow } },
})
}
}(script)
}
it('records capacity once and skips it while the route is unchanged', async () => {
const adapter = capacityAdapter({ mock: 128_000 }, [textResponse('a'), textResponse('b')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('capacity-dedup'), { provider: 'mock', model: 'mock' })
send(agent, 'first')
await waitForIdle(ctx, agent)
send(agent, 'second')
await waitForIdle(ctx, agent)
const records = agent.session.events.filter(event => event.type === 'request/context')
expect(records).toHaveLength(1)
expect(records[0]?.data).toEqual({ provider: 'mock', model: 'mock', contextWindow: 128_000 })
// Log-only: not a SurfaceEventType, so it can never reach a model request
// (the type system rejects a surfaceOp here; the session invariant also
// requires the record to sit inside its open turn).
expect(agent.session.surface.nodes).not.toContain(records[0]?.seq)
})
it('records a second capacity when the route changes mid-session', async () => {
const adapter = capacityAdapter(
{ small: 64_000, large: 256_000 },
[textResponse('a'), textResponse('b')],
)
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('capacity-switch'), { provider: 'mock', model: 'small' })
send(agent, 'first')
await waitForIdle(ctx, agent)
ctx.on('agent/request', (subject, _turn, _step, _signal, next) => subject === agent
? Promise.resolve({ provider: 'mock', model: 'large' })
: next())
send(agent, 'second')
await waitForIdle(ctx, agent)
expect(agent.session.events
.filter(event => event.type === 'request/context')
.map(event => event.data.contextWindow)).toEqual([64_000, 256_000])
})
it('records and deduplicates a route whose adapter advertises no capacity', async () => {
const ctx = await harness(new MockAdapter([textResponse('a'), textResponse('b')]))
const agent = ctx.agentLoop.create(SessionId('capacity-absent'), { provider: 'mock', model: 'mock' })
send(agent, 'first')
await waitForIdle(ctx, agent)
send(agent, 'second')
await waitForIdle(ctx, agent)
expect(agent.session.events
.filter(event => event.type === 'request/context')
.map(event => event.data)).toEqual([{ provider: 'mock', model: 'mock' }])
})
it('clears a previous capacity when the next route advertises none', async () => {
const adapter = capacityAdapter({ known: 64_000 }, [textResponse('a'), textResponse('b')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('capacity-clear'), { provider: 'mock', model: 'known' })
let model = 'known'
ctx.on('agent/request', (subject, _turn, _step, _signal, next) => subject === agent
? Promise.resolve({ provider: 'mock', model })
: next())
send(agent, 'first')
await waitForIdle(ctx, agent)
model = 'unknown'
send(agent, 'second')
await waitForIdle(ctx, agent)
expect(agent.session.events
.filter(event => event.type === 'request/context')
.map(event => event.data)).toEqual([
{ provider: 'mock', model: 'known', contextWindow: 64_000 },
{ provider: 'mock', model: 'unknown' },
])
})
})

View File

@@ -0,0 +1,286 @@
import { describe, expect, it } from 'vitest'
import { Context } from 'cordis'
import AgentRegistry, { type Agent, type InboxItem } from '@deepseek-ai/dsh-agent'
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
import LlmService, { createUserMessage } from '@deepseek-ai/dsh-llm'
import SessionStore, { SessionId } from '@deepseek-ai/dsh-session'
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
import ToolRegistry from '@deepseek-ai/dsh-tools'
import { MockAdapter, textResponse } from './mock-adapter.ts'
async function harness(adapter: MockAdapter): Promise<Context> {
const ctx = new Context()
await ctx.plugin(LlmService)
await ctx.plugin(SessionStore)
await ctx.plugin(SystemPrompt)
await ctx.plugin(ToolRegistry)
await ctx.plugin(AgentRegistry)
await ctx.plugin(AgentLoop, { agents: [] })
ctx.llm.registerAdapter(['mock'], adapter)
return ctx
}
function prompt(agent: Agent, text: string): void {
agent.followup(createUserMessage({
content: [{ type: 'text', text }],
source: { kind: 'user' },
}))
}
function itemText(item: InboxItem): string {
return item.message.content.flatMap(block => block.type === 'text' ? [block.text] : []).join('')
}
interface InboxRecording {
readonly events: string[]
readonly enqueued: InboxItem['id'][]
readonly dequeued: InboxItem['id'][]
readonly discarded: InboxItem['id'][]
}
/** Record the complete inbox lifecycle of one agent for order and identity assertions. */
function recordInbox(ctx: Context): InboxRecording {
const events: string[] = []
const enqueued: InboxItem['id'][] = []
const dequeued: InboxItem['id'][] = []
const discarded: InboxItem['id'][] = []
ctx.on('agent/inbox/enqueue', (_agent, item) => {
events.push(`enqueue:${item.placement}:${itemText(item)}`)
enqueued.push(item.id)
})
ctx.on('agent/inbox/dequeue', (_agent, item) => {
events.push(`dequeue:${itemText(item)}`)
dequeued.push(item.id)
})
ctx.on('agent/inbox/discard', (_agent, items) => {
events.push(`discard:${items.map(itemText).join(',')}`)
discarded.push(...items.map(item => item.id))
})
return { events, enqueued, dequeued, discarded }
}
/** Text of every ordinary prompt the log admitted, in durable order. */
function promptTexts(agent: Agent): string[] {
return agent.session.events.flatMap(event => event.type === 'user/message'
? event.data.content.flatMap(block => block.type === 'text' ? [block.text] : [])
: [])
}
describe('idle turn admission reservation', () => {
it('holds later waking prompts in the FIFO until release', async () => {
const adapter = new MockAdapter([textResponse('first'), textResponse('second')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const inbox = recordInbox(ctx)
const release = agent.reserveTurnAdmission()
expect(release).toBeDefined()
prompt(agent, 'first prompt')
prompt(agent, 'second prompt')
expect(agent.acceptsNextStep).toBe(false)
await new Promise<void>((resolve) => { setTimeout(resolve, 5) })
expect(agent.status).toBe('idle')
expect(adapter.requests).toHaveLength(0)
expect(agent.session.events).toHaveLength(0)
expect(inbox.events).toEqual([
'enqueue:queued:first prompt',
'enqueue:queued:second prompt',
])
release?.()
await agent.whenIdle()
expect(promptTexts(agent)).toEqual(['first prompt', 'second prompt'])
expect(agent.session.events.flatMap(event =>
event.type === 'turn/start' ? [event.data.turn] : [])).toEqual([1, 2])
expect(inbox.events).toEqual([
'enqueue:queued:first prompt',
'enqueue:queued:second prompt',
'dequeue:first prompt',
'dequeue:second prompt',
])
expect(inbox.dequeued).toEqual(inbox.enqueued)
expect(inbox.discarded).toEqual([])
})
it('refuses acquisition when an accepted waking prompt still owns the next turn', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
prompt(agent, 'accepted first')
expect(agent.status).toBe('idle')
expect(agent.reserveTurnAdmission()).toBeUndefined()
await agent.whenIdle()
expect(adapter.requests).toHaveLength(1)
})
it('refuses acquisition while a turn is running', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const reserved: unknown[] = []
ctx.on('agent/step', () => {
reserved.push(agent.reserveTurnAdmission())
})
prompt(agent, 'running')
await agent.whenIdle()
expect(agent.status).toBe('idle')
expect(reserved).toEqual([undefined])
})
it('refuses a second reservation and releases idempotently', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const release = agent.reserveTurnAdmission()
expect(agent.reserveTurnAdmission()).toBeUndefined()
prompt(agent, 'queued behind the reservation')
release?.()
release?.()
await agent.whenIdle()
expect(promptTexts(agent)).toEqual(['queued behind the reservation'])
expect(adapter.requests).toHaveLength(1)
const second = agent.reserveTurnAdmission()
expect(second).toBeDefined()
second?.()
})
it('ignores a stale release once a later reservation owns the boundary', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const stale = agent.reserveTurnAdmission()
stale?.()
const live = agent.reserveTurnAdmission()
prompt(agent, 'held by the live reservation')
stale?.()
await new Promise<void>((resolve) => { setTimeout(resolve, 5) })
expect(adapter.requests).toHaveLength(0)
live?.()
await agent.whenIdle()
expect(adapter.requests).toHaveLength(1)
})
it('acquires beside quiet queued work and leaves it queued', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
agent.send(createUserMessage({
content: [{ type: 'text', text: 'quiet' }],
source: { kind: 'user' },
}), {
target: 'next-turn',
wakeup: false,
})
const release = agent.reserveTurnAdmission()
expect(release).toBeDefined()
release?.()
await agent.whenIdle()
expect(adapter.requests).toHaveLength(0)
})
it('makes whenIdle() wait for release without spinning on a settled promise', async () => {
const adapter = new MockAdapter([textResponse('ok')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const machine = agent as Agent & { done: Promise<void> }
let backing = machine.done
let reads = 0
Object.defineProperty(agent, 'done', {
configurable: true,
get(): Promise<void> {
reads += 1
return backing
},
set(value: Promise<void>) {
backing = value
},
})
const release = agent.reserveTurnAdmission()
prompt(agent, 'waiting for the reservation')
let settled = false
const idle = agent.whenIdle().then(() => { settled = true })
for (let tick = 0; tick < 5; tick += 1) {
await new Promise<void>((resolve) => { setTimeout(resolve, 1) })
}
expect(settled).toBe(false)
expect(reads).toBeLessThanOrEqual(2)
release?.()
await idle
expect(settled).toBe(true)
expect(adapter.requests).toHaveLength(1)
})
it('resolves whenIdle() after release with nothing queued', async () => {
const adapter = new MockAdapter([])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const release = agent.reserveTurnAdmission()
let settled = false
const idle = agent.whenIdle().then(() => { settled = true })
await new Promise<void>((resolve) => { setTimeout(resolve, 5) })
expect(settled).toBe(false)
release?.()
await idle
expect(agent.status).toBe('idle')
})
it('lets cancellation discard held prompts and keeps the boundary quiet', async () => {
const adapter = new MockAdapter([])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
const inbox = recordInbox(ctx)
const release = agent.reserveTurnAdmission()
prompt(agent, 'discarded while held')
agent.cancel({ kind: 'user' })
expect(inbox.events).toEqual([
'enqueue:queued:discarded while held',
'discard:discarded while held',
])
expect(inbox.discarded).toEqual(inbox.enqueued)
expect(inbox.dequeued).toEqual([])
release?.()
await agent.whenIdle()
expect(adapter.requests).toHaveLength(0)
expect(agent.session.events).toHaveLength(0)
})
it('disposes the agent without waiting for the reservation to be released', async () => {
const adapter = new MockAdapter([])
const ctx = await harness(adapter)
const handle = await ctx.agents.create({
sessionId: SessionId('a1'),
agentOptions: { provider: 'mock', model: 'mock' },
})
const { agent } = handle
const release = agent.reserveTurnAdmission()
prompt(agent, 'discarded by disposal')
await handle.dispose()
expect(ctx.agents.list()).toEqual([])
expect(adapter.requests).toHaveLength(0)
release?.()
})
})

View File

@@ -29,6 +29,7 @@ function stubAgent(rawId: string, overrides: Partial<Agent> = {}): Agent {
followup: () => {},
steer: () => {},
inject: () => {},
reserveTurnAdmission: () => undefined,
cancel() {},
whenIdle: () => Promise.resolve(),
}

View File

@@ -67,7 +67,7 @@ describe('SessionStore.fork', () => {
const child = sessions.fork(source, undefined, SessionId('empty-child'))
expect(child.events).toEqual([])
expect(inherited(child)).toEqual([])
expect(child.header).toMatchObject({
id: SessionId('empty-child'),
cwd: '/workspace',

View File

@@ -143,6 +143,12 @@ describe('session-log invariants', () => {
source: { kind: 'user' },
}),
}, { surfaceOp: 'append' })).toThrow(/outside any open turn/)
// Route capacity is core execution state like the header beside it.
expect(() => outside.append('request/context', {
provider: 'mock',
model: 'm',
contextWindow: 128_000,
})).toThrow(/outside any open turn/)
// The owning plugin decides whether a merge-extensible event is log-only.
const appendUnknown = outside.append.bind(outside) as (type: string, data: unknown) => unknown
expect(() => { appendUnknown('plugin/marker', {}) }).not.toThrow()

View File

@@ -112,9 +112,9 @@ describe('Session properties', () => {
const original = build(events)
const replayed = new Session(SessionId(`replay-${counter++}`), [...original.events])
expect(replayed.deriveMessages()).toEqual(original.deriveMessages())
// A non-empty replay grows by exactly one log-only boundary.
// Every explicit replay grows by exactly one log-only boundary.
expect(replayed.events.slice(0, original.seq)).toEqual(original.events)
expect(replayed.seq).toBe(original.seq === 0 ? 0 : original.seq + 1)
expect(replayed.seq).toBe(original.seq + 1)
}))
})

View File

@@ -114,3 +114,60 @@ describe('legacy request-header format', () => {
expect(session.events).toHaveLength(0)
})
})
describe('Session.requestContext', () => {
const CAPACITY = { provider: 'mock', model: 'm', contextWindow: 128_000 }
/** A turn-enclosed capacity record; the invariant rejects one outside a turn. */
function seedWith(...records: { provider: string; model: string; contextWindow?: number }[]): SessionEvent[] {
const events: SessionEvent[] = [{
type: 'turn/start', seq: 0, time: 1, data: { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } },
}]
for (const data of records) {
events.push({ type: 'request/context', seq: events.length, time: 1, data })
}
return events
}
it('reads undefined before any record exists', () => {
expect(new Session(SessionId('no-capacity')).requestContext()).toBeUndefined()
})
it('folds a seeded log on first read, taking the last record', () => {
// The fold watermark starts at 0 with the seed already in the log, so the
// first read must consume the whole seed rather than skip it.
const session = new Session(SessionId('seeded-capacity'), seedWith(
CAPACITY,
{ ...CAPACITY, model: 'later', contextWindow: 256_000 },
))
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'later', contextWindow: 256_000 })
})
it('advances incrementally across appends and skips unrelated events', () => {
const session = new Session(SessionId('incremental-capacity'), seedWith(CAPACITY))
expect(session.requestContext()).toEqual(CAPACITY)
session.append('todo/write', { todos: [] })
expect(session.requestContext()).toEqual(CAPACITY)
session.append('request/context', { ...CAPACITY, model: 'next', contextWindow: 64_000 })
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'next', contextWindow: 64_000 })
session.append('request/context', { provider: 'mock', model: 'unknown' })
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'unknown' })
})
it('folds a batch appended between two reads', () => {
const session = new Session(SessionId('batched-capacity'), seedWith(CAPACITY))
expect(session.requestContext()).toEqual(CAPACITY)
session.append('request/context', { ...CAPACITY, contextWindow: 200_000 })
session.append('todo/write', { todos: [] })
session.append('request/context', { ...CAPACITY, contextWindow: 300_000 })
expect(session.requestContext()?.contextWindow).toBe(300_000)
})
it('exposes a frozen record so a reader cannot desync later comparisons', () => {
const session = new Session(SessionId('frozen-capacity'), seedWith(CAPACITY))
const held = session.requestContext()
if (held === undefined) throw new Error('expected a folded capacity record')
expect(Object.isFrozen(held)).toBe(true)
expect(() => { (held as { contextWindow?: number }).contextWindow = 1 }).toThrow()
})
})

View File

@@ -142,6 +142,21 @@ describe('Session', () => {
expect(replayed.firstLiveSeq).toBe(original.seq)
})
it('marks an explicitly empty seed without marking a fresh session', () => {
const fresh = new Session(SessionId('fresh-empty'))
expect(fresh.events).toEqual([])
const resumed = new Session(SessionId('resumed-empty'), [])
expect(resumed.firstLiveSeq).toBe(0)
expect(resumed.events).toMatchObject([
{ type: 'session/end-seed', seq: 0, data: {} },
])
const reopened = new Session(SessionId('reopened-empty'), resumed.events)
expect(reopened.firstLiveSeq).toBe(1)
expect(reopened.events).toEqual(resumed.events)
})
it('rejects pre-provider request headers and assistant messages on seed/load', () => {
const requestHeader = {
type: 'request/header', seq: 0, time: 1,

View File

@@ -13,6 +13,7 @@ async function setup(): Promise<Context> {
const valid = (): PromptAssembly => ({
sections: [{ name: 'identity', text: 'prompt' }],
contexts: [{ name: 'policy', text: 'current policy' }],
tools: [{ name: 'echo', description: 'Echo', parameters: {} }],
variables: { cwd: '/repo', optional: undefined },
})
@@ -34,6 +35,9 @@ describe('system-prompt invariants', () => {
[{ ...valid(), sections: [{ name: '', text: 'x' }] }, /section names must be non-empty/],
[{ ...valid(), sections: [{ name: 'x', text: 'a' }, { name: 'x', text: 'b' }] }, /section name "x" is duplicated/],
[{ ...valid(), sections: [{ name: 'x', text: 1 as never }] }, /section "x" text must be a string/],
[{ ...valid(), contexts: [{ name: '', text: 'x' }] }, /context names must be non-empty/],
[{ ...valid(), contexts: [{ name: 'x', text: 'a' }, { name: 'x', text: 'b' }] }, /context name "x" is duplicated/],
[{ ...valid(), contexts: [{ name: 'x', text: 1 as never }] }, /context "x" text must be a string/],
[{ ...valid(), tools: [{ name: '', description: 'x', parameters: {} }] }, /tool names must be non-empty/],
[{ ...valid(), variables: { Bad: 'x' } }, /variable name "Bad" is invalid/],
[{ ...valid(), variables: { value: 1 as never } }, /variable "value" must be a string or undefined/],

View File

@@ -2,7 +2,7 @@ import { describe, expect, it, vi } from 'vitest'
import { Context } from 'cordis'
import { createScope, scopeOf } from '@deepseek-ai/dsh-scope'
import type { Scope, ScopeKey } from '@deepseek-ai/dsh-scope'
import SystemPrompt, { TOOL_ORDER_REST, renderPrompt } from '@deepseek-ai/dsh-system-prompt'
import SystemPrompt, { TOOL_ORDER_REST, renderContextSnapshot, renderPrompt } from '@deepseek-ai/dsh-system-prompt'
import type { Config, PromptAssembly } from '@deepseek-ai/dsh-system-prompt'
async function mount(config: Config = {}): Promise<Context> {
@@ -125,6 +125,25 @@ describe('scoped variables', () => {
})
})
describe('scoped cache-safe context', () => {
it('shadows a global context for one scope and cleans up with that scope', async () => {
const ctx = await mount()
const scope = await mintScope(ctx, 'child-context')
ctx.systemPrompt.context({ name: 'policy', order: 1, text: 'global policy' })
scope.ctx.systemPrompt.context({ name: 'policy', order: 1, text: 'scoped policy' })
expect(() => scope.ctx.systemPrompt.context({ name: 'policy', order: 2, text: 'duplicate' }))
.toThrow('prompt context "policy" is already registered in this scope')
expect(renderContextSnapshot(await ctx.systemPrompt.assemble({ scope: scopeKeyOf(scope) })))
.toContain('scoped policy')
expect(renderContextSnapshot(await ctx.systemPrompt.assemble())).toContain('global policy')
await scope.dispose()
expect(renderContextSnapshot(await ctx.systemPrompt.assemble({ scope: scopeKeyOf(scope) })))
.toContain('global policy')
})
})
describe('scoped tool providers and toolOrder × restriction', () => {
it('scoped providers are consulted only for their scope', async () => {
const ctx = await mount()

View File

@@ -1,6 +1,6 @@
import { describe, expect, it } from 'vitest'
import { Context } from 'cordis'
import SystemPrompt, { AssembleContext, PromptAssembly, renderPrompt } from '@deepseek-ai/dsh-system-prompt'
import SystemPrompt, { AssembleContext, PromptAssembly, renderContextSnapshot, renderPrompt } from '@deepseek-ai/dsh-system-prompt'
/**
* Every assembly carries the plugin's own built-ins — `harness:identity`
@@ -64,14 +64,21 @@ describe('SystemPrompt', () => {
ctx.systemPrompt.section({ name: 'cwd', order: 20, text: () => 'cwd: /tmp' })
ctx.systemPrompt.section({ name: 'rules', order: 10, text: 'Be precise.' })
ctx.systemPrompt.context({ name: 'later', order: 20, text: () => 'context 2' })
ctx.systemPrompt.context({ name: 'earlier', order: 10, text: 'context 1' })
ctx.systemPrompt.tools(() => ({ schemas: [{ name: 'echo', description: 'echo back', parameters: {} }] }))
const assembly = await ctx.systemPrompt.assemble()
expect(assembly.sections.map(s => s.name)).toEqual(['harness:identity', 'deployment:persona', 'rules', 'cwd'])
expect(assembly.sections.map(s => s.text)).toEqual([IDENTITY, 'You are DeepSeek Harness SDK.', 'Be precise.', 'cwd: /tmp'])
expect(assembly.contexts).toEqual([
{ name: 'earlier', text: 'context 1' },
{ name: 'later', text: 'context 2' },
])
expect(assembly.tools).toEqual([{ name: 'echo', description: 'echo back', parameters: {} }])
expect(assembly.variables).toEqual({})
expect(renderPrompt(assembly)).toBe(`${IDENTITY}\n\nYou are DeepSeek Harness SDK.\n\nBe precise.\n\ncwd: /tmp`)
expect(renderContextSnapshot(assembly)).toBe('Current runtime context. This snapshot supersedes earlier runtime-context snapshots.\n\ncontext 1\n\ncontext 2')
})
it('resolves section text providers against the assemble context, at each assemble call', async () => {
@@ -96,16 +103,19 @@ describe('SystemPrompt', () => {
const fiber = await ctx.plugin(Object.assign((inner: Context) => {
inner.systemPrompt.section({ name: 'scoped', order: 0, text: 'scoped section' })
inner.systemPrompt.context({ name: 'scoped-context', order: 0, text: 'scoped context' })
inner.systemPrompt.tools(() => ({ schemas: [{ name: 'scoped-tool', description: '', parameters: {} }] }))
inner.systemPrompt.variable('scoped_var', () => 'v')
}, { inject: ['systemPrompt'] }))
const before = await ctx.systemPrompt.assemble()
expect(contributed(before)).toHaveLength(1)
expect(before.contexts).toHaveLength(1)
expect(before.variables).toEqual({ scoped_var: 'v' })
await fiber.dispose()
const assembly = await ctx.systemPrompt.assemble()
expect(contributed(assembly)).toHaveLength(0)
expect(assembly.contexts).toHaveLength(0)
// The built-ins belong to the service fiber, so they survive the plugin's disposal.
expect(assembly.sections.map(s => s.name)).toEqual(BUILT_IN)
expect(assembly.tools).toHaveLength(0)
@@ -131,6 +141,17 @@ describe('SystemPrompt', () => {
expect(contributed(await ctx.systemPrompt.assemble())).toEqual([])
})
it('rejects duplicate and non-finite context registrations without leaking', async () => {
const ctx = new Context()
await ctx.plugin(SystemPrompt)
ctx.systemPrompt.context({ name: 'policy', order: 1, text: 'first' })
expect(() => ctx.systemPrompt.context({ name: 'policy', order: 2, text: 'second' }))
.toThrow('prompt context "policy" is already registered')
expect(() => ctx.systemPrompt.context({ name: 'bad', order: Number.NaN, text: 'x' }))
.toThrow('prompt context "bad" order must be a finite number')
expect((await ctx.systemPrompt.assemble()).contexts).toEqual([{ name: 'policy', text: 'first' }])
})
it('rolls back a section when a system-prompt/change listener throws (P1-1)', async () => {
const ctx = new Context()
await ctx.plugin(SystemPrompt)
@@ -236,7 +257,7 @@ describe('SystemPrompt', () => {
ctx.systemPrompt.section({ name: 'real', order: 0, text: 'real' })
ctx.on('system-prompt/assemble', async () => {
return { sections: [], tools: [], variables: {} } satisfies PromptAssembly
return { sections: [], contexts: [], tools: [], variables: {} } satisfies PromptAssembly
})
const assembly = await ctx.systemPrompt.assemble()
@@ -252,6 +273,7 @@ describe('SystemPrompt', () => {
const first = await ctx.systemPrompt.assemble()
first.sections[0]!.name = 'mutated'
first.sections[0]!.text = 'mutated'
first.contexts.push({ name: 'mutated', text: 'mutated' })
first.tools[0]!.description = 'mutated'
const firstParameters = first.tools[0]!.parameters as { properties: Record<string, unknown> }
firstParameters.properties['leak'] = { type: 'string' }
@@ -259,6 +281,7 @@ describe('SystemPrompt', () => {
const second = await ctx.systemPrompt.assemble()
expect(second.sections.map(section => section.name)).toEqual(['harness:identity', 'deployment:persona', 'base'])
expect(second.sections[0]!.text).toBe(IDENTITY)
expect(second.contexts).toEqual([])
expect(second.tools).toEqual([{ name: 't', description: 'tool', parameters: { type: 'object', properties: {} } }])
})
@@ -268,12 +291,33 @@ describe('SystemPrompt', () => {
{ name: 'empty', text: '' },
{ name: 'real', text: 'content' },
],
contexts: [],
tools: [],
variables: {},
})
expect(result).toBe('content')
})
it('filters empty context, interpolates variables, and returns empty without active context', async () => {
const ctx = new Context()
await ctx.plugin(SystemPrompt)
ctx.systemPrompt.context({ name: 'empty', order: 0, text: '' })
expect(renderContextSnapshot(await ctx.systemPrompt.assemble())).toBe('')
ctx.systemPrompt.variable('mode', () => 'read-only')
ctx.systemPrompt.context({ name: 'policy', order: 1, text: 'Mode: {{mode}}.' })
expect(renderContextSnapshot(await ctx.systemPrompt.assemble()))
.toBe('Current runtime context. This snapshot supersedes earlier runtime-context snapshots.\n\nMode: read-only.')
})
it('attributes context interpolation failures to the contributing context', () => {
expect(() => renderContextSnapshot({
sections: [],
contexts: [{ name: 'policy', text: 'Mode: {{missing}}.' }],
tools: [],
variables: {},
})).toThrow('unknown prompt variable "{{missing}}" in context "policy"; registered variables: (none)')
})
it('emits system-prompt/change when a tool provider is registered and disposed', async () => {
const ctx = new Context()
await ctx.plugin(SystemPrompt)
@@ -290,6 +334,17 @@ describe('SystemPrompt', () => {
expect(changeCount).toBe(2)
})
it('emits system-prompt/change when a context is registered and disposed', async () => {
const ctx = new Context()
await ctx.plugin(SystemPrompt)
let changeCount = 0
ctx.on('system-prompt/change', () => void changeCount++)
const dispose = ctx.systemPrompt.context({ name: 'policy', order: 0, text: 'current' })
expect(changeCount).toBe(1)
dispose()
expect(changeCount).toBe(2)
})
it('cleans up tool providers on fiber dispose', async () => {
const ctx = new Context()
await ctx.plugin(SystemPrompt)
@@ -404,13 +459,14 @@ describe('SystemPrompt', () => {
})
it('names "(none)" when no variables are registered at all', () => {
expect(() => renderPrompt({ sections: [{ name: 's', text: '{{x}}' }], tools: [], variables: {} }))
expect(() => renderPrompt({ sections: [{ name: 's', text: '{{x}}' }], contexts: [], tools: [], variables: {} }))
.toThrow('unknown prompt variable "{{x}}" in section "s"; registered variables: (none)')
})
it('throws when a referenced variable has no value for this assembly', () => {
expect(() => renderPrompt({
sections: [{ name: 'persona', text: 'in {{cwd}}' }],
contexts: [],
tools: [],
variables: { cwd: undefined },
})).toThrow('prompt variable "{{cwd}}" has no value for this assembly (section "persona")')
@@ -419,6 +475,7 @@ describe('SystemPrompt', () => {
it('throws on a malformed complete reference, e.g. inner spaces', () => {
expect(() => renderPrompt({
sections: [{ name: 's', text: 'on {{ model }}' }],
contexts: [],
tools: [],
variables: { model: 'm' },
})).toThrow('malformed prompt variable reference "{{ model }}" in section "s"')
@@ -427,6 +484,7 @@ describe('SystemPrompt', () => {
it('leaves a lone {{ verbatim only when NO }} follows anywhere after it', () => {
const text = renderPrompt({
sections: [{ name: 's', text: 'shell ${X:-{{fallback} stays' }],
contexts: [],
tools: [],
variables: {},
})
@@ -439,6 +497,7 @@ describe('SystemPrompt', () => {
])('throws on a mangled reference with a }} still following ($label)', ({ text }) => {
expect(() => renderPrompt({
sections: [{ name: 's', text }],
contexts: [],
tools: [],
variables: { model: 'm' },
})).toThrow('malformed prompt variable reference at')
@@ -449,6 +508,7 @@ describe('SystemPrompt', () => {
// source into the prompt; Object.hasOwn must reject it instead.
expect(() => renderPrompt({
sections: [{ name: 's', text: 'on {{constructor}}' }],
contexts: [],
tools: [],
variables: { model: 'm' },
})).toThrow('unknown prompt variable "{{constructor}}"')
@@ -465,6 +525,7 @@ describe('SystemPrompt', () => {
it('never re-scans substituted values (a value containing {{sneaky}} stays literal)', () => {
const text = renderPrompt({
sections: [{ name: 's', text: 'v = {{model}}!' }],
contexts: [],
tools: [],
variables: { model: 'literal {{sneaky}} inside' },
})

View File

@@ -2,5 +2,5 @@
# side as of the last confirmed-consistent state. Both languages carry equal authority;
# after editing either side, bring the other along and re-record with:
# pnpm run verify-translation-pairing --write packages/core/tools/README.md
README.md: e7f395f8c1d6417db856e590f5267cf6887e4d12
README.zh.md: acb4c047bf86e36c828882ff751d4be1f627f99e
README.md: 15fc5839a3b0e3fa2d20c5a9cc50577e9807ffda
README.zh.md: 8547ee4a796dcd93945dfa40373c14c10d7d0c8a

View File

@@ -108,7 +108,7 @@ Optional `isConcurrencySafe(args)` receives typed, softly validated arguments. E
Tools optionally own pure `presentCall()` and `presentResult()` render intents, so UIs do not special-case tool names:
- Call views are `{ card: 'generic', title, kind?, rawInput?, content?, locations? }`, `{ card: 'terminal', title, description?, cwd? }`, or `{ card: 'diff', title, diffs, locations? }`.
- Result views are `{ card: 'generic', title?, content? }`, `{ card: 'terminal', title?, output?, exitCode?, signal? }`, `{ card: 'diff', title?, diffs }`, or `{ card: 'web', kind: 'search' | 'fetch', title?, … }` (a completed web retrieval; the `kind` arms carry the structured search sources or the fetch summary, and a UI without the `web` capability falls back to the raw result content).
- Result views are `{ card: 'generic', title?, content? }`, `{ card: 'terminal', title?, output?, exitCode?, signal? }`, `{ card: 'diff', title?, diffs }`, `{ card: 'search', shape, title?, truncated, total, … }` (a completed discovery search — grouped-by-file matches for `shape: 'matches'` (grep) or a flat path list for `shape: 'paths'` (glob), with `truncated`/`total` so a UI never presents a capped result as complete; the view carries no result text and a search has no `card: 'search'` call-time analogue), `{ card: 'read', title?, path, offset, lines, totalLines, lang?, content? }` (a completed file read → a line-numbered, optionally syntax-highlighted code view; `offset` is the 1-based first line the window requested, kept even when `lines` is empty; `lines` is `{ number, text }[]` keeping each file line number, and `content` is the envelope-stripped text a UI without read support falls back to), or `{ card: 'web', kind: 'search' | 'fetch', title?, … }` (a completed web retrieval; the `kind` arms carry the structured search sources or the fetch summary, and a UI without the `web` capability falls back to the raw result content).
Returning `undefined` selects generic fallback. Presenters depend only on their arguments and the durable result because UIs call them during live streaming and log replay. `output.presentationMeta(args, value)` derives JSON metadata for direct surface calls; that metadata persists with `tool/result` and returns to `presentResult`, while the canonical value itself remains execution-local and is never replayed. Nested Code dispatches do not compute metadata. `defineTool` soft-validates older logged arguments and falls back instead of crashing replay. `dsh-tool-bash` and `dsh-tool-fs` are the reference implementations; the [canonical-output Agent Note](../../../.agents/notes/implemented/architecture/2026-07-20-canonical-tool-output-contract.md) owns the value/presentation split and the [render-intent Agent Note](../../../.agents/notes/implemented/architecture/2026-07-02-tool-render-intent-union.md) owns card vocabulary.

View File

@@ -108,7 +108,7 @@ ctx.tools.register(defineTool({
工具可以选择拥有纯 `presentCall()` 和 `presentResult()` 呈现意图,使 UI 无需特殊处理工具名称:
- 调用视图为 `{ card: 'generic', title, kind?, rawInput?, content?, locations? }`、`{ card: 'terminal', title, description?, cwd? }` 或 `{ card: 'diff', title, diffs, locations? }`。
- 结果视图为 `{ card: 'generic', title?, content? }`、`{ card: 'terminal', title?, output?, exitCode?, signal? }`、`{ card: 'diff', title?, diffs }` 或 `{ card: 'web', kind: 'search' | 'fetch', title?, … }`(已完成的 web 检索;`kind` 各分支携带结构化的搜索来源或抓取摘要,不具备 `web` 能力的 UI 回退到原始结果内容)。
- 结果视图为 `{ card: 'generic', title?, content? }`、`{ card: 'terminal', title?, output?, exitCode?, signal? }`、`{ card: 'diff', title?, diffs }`、`{ card: 'search', shape, title?, truncated, total, … }`(已完成的发现型搜索——`shape: 'matches'`(grep)为按文件分组的匹配,`shape: 'paths'`(glob)为扁平路径列表,配 `truncated`/`total` 使 UI 永不把被截断的结果当作完整结果呈现;该视图不携带结果文本,且搜索没有 `card: 'search'` 的调用时对应视图)、`{ card: 'read', title?, path, offset, lines, totalLines, lang?, content? }`(已完成的文件读取→带行号、可选语法高亮的代码视图;`offset` 是窗口请求的 1-based 起始行,即使 `lines` 为空也保留;`lines` 是 `{ number, text }[]`,保留每一行的文件行号,`content` 是无读取能力的 UI 回退时使用的去信封文本)或 `{ card: 'web', kind: 'search' | 'fetch', title?, … }`(已完成的 web 检索;`kind` 各分支携带结构化的搜索来源或抓取摘要,不具备 `web` 能力的 UI 回退到原始结果内容)。
返回 `undefined` 会选择通用回退。呈现器只依赖其参数和持久结果,因为 UI 会在实时流式输出和日志回放期间调用它们。`output.presentationMeta(args, value)` 为直接接口调用派生 JSON 元数据;该元数据随 `tool/result` 持久化并传回 `presentResult`,而规范值本身仍只存在于执行局部,绝不会回放。嵌套 Code 分发不会计算元数据。`defineTool` 会软验证较旧的日志参数并回退,而不会使回放崩溃。`dsh-tool-bash` 与 `dsh-tool-fs` 是参考实现;[规范输出 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-20-canonical-tool-output-contract.md) 规定值/呈现拆分,[呈现意图 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-02-tool-render-intent-union.md) 规定卡片词汇。

View File

@@ -74,6 +74,7 @@ export type {
ToolCallKind,
FileLocation,
FileDiff,
ReadFileLine,
ToolCallView,
GenericCallView,
TerminalCallView,
@@ -82,6 +83,12 @@ export type {
GenericResultView,
TerminalResultView,
DiffResultView,
SearchResultView,
SearchMatchesResultView,
SearchPathsResultView,
SearchFileMatches,
SearchLineMatch,
ReadResultView,
WebResultView,
WebSearchResultView,
WebFetchResultView,

View File

@@ -117,6 +117,18 @@ export interface DiffCallView {
locations?: FileLocation[]
}
/**
* One numbered line of a file, the unit a {@link ReadResultView} carries so a
* capable UI can render a syntax-highlighted, line-numbered code view. `number`
* is the 1-based line number in the file (a window past `offset` keeps the file's
* own numbering, not a 1-based re-count); `text` is the line without its trailing
* newline, already truncated to the read tool's per-line cap.
*/
export interface ReadFileLine {
number: number
text: string
}
/**
* How a tool wants the COMPLETED call shown — the *result* state, after `execute`
* returns. A `card`-tagged union mirroring {@link ToolCallView}: a UI switches on
@@ -125,7 +137,7 @@ export interface DiffCallView {
* `ToolDefinition.presentResult`; omitting the method keeps the pending
* title and renders the raw result content.
*/
export type ToolResultView = GenericResultView | TerminalResultView | DiffResultView | WebResultView
export type ToolResultView = GenericResultView | TerminalResultView | DiffResultView | SearchResultView | ReadResultView | WebResultView
/**
* The default completed card: an optional replacement title and reformatted
@@ -177,6 +189,124 @@ export interface DiffResultView {
diffs: FileDiff[]
}
/** One matched line inside a {@link SearchFileMatches} group: its 1-based line number and text. */
export interface SearchLineMatch {
/** 1-based line number of the match within its file. */
lineNumber: number
/** The matched line text, as the tool surfaced it (the per-line preview budget already applied). */
line: string
}
/** One file's grouped content matches for a {@link SearchMatchesResultView}, in first-seen file order. */
export interface SearchFileMatches {
/** The file the matches belong to (the model-facing display path). */
path: string
/** The file's matched lines, in output order. */
matches: SearchLineMatch[]
}
/**
* A completed content search (`grep`) rendered as a search card whose matches are
* grouped by file, so a capable UI can list each file as an expandable group of
* its matched lines. `shape: 'matches'` discriminates this variant from the path
* variant ({@link SearchPathsResultView}) within {@link SearchResultView}. The
* discriminant is `shape`, not `kind`, so it never collides with the
* {@link ToolCallKind} `kind` an icon-picking bridge reads off a call view.
*/
export interface SearchMatchesResultView {
card: 'search'
shape: 'matches'
/** Replacement title for the completed call. Omit to keep the pending-state title. */
title?: string
/** Matched lines grouped by file, in first-seen file order. */
files: SearchFileMatches[]
/**
* Whether the tool capped the inline result: `files` carries only the retained
* matches, not every match the search found. A UI shows a capped indicator so it
* never presents a partial group as complete.
*/
truncated: boolean
/** Total matches the search found before capping (equals the retained count when not `truncated`). */
total: number
}
/**
* A completed path search (`glob`) rendered as a search card whose result is a flat
* path list. `shape: 'paths'` discriminates this variant from the grouped-matches
* variant ({@link SearchMatchesResultView}) within {@link SearchResultView}.
*/
export interface SearchPathsResultView {
card: 'search'
shape: 'paths'
/** Replacement title for the completed call. Omit to keep the pending-state title. */
title?: string
/** The discovered paths, in the tool's result order (the retained page when `truncated`). */
paths: string[]
/**
* Whether the tool capped the inline result: `paths` carries only the retained
* page, not every path the search found. A UI shows a capped indicator so it
* never presents a partial list as complete.
*/
truncated: boolean
/** Total paths the search found before capping (equals `paths.length` when not `truncated`). */
total: number
}
/**
* A completed search rendered as a search card, the result-time view a discovery
* tool (`grep`, `glob`) returns from `presentResult`. One `card: 'search'` view
* with two `shape`-discriminated variants: grouped-by-file content matches
* ({@link SearchMatchesResultView}) and a flat path list
* ({@link SearchPathsResultView}). Both carry a `truncated`/`total` signal so a UI
* never presents a capped result as complete. The view carries no result text: a
* UI without a search card falls back to the raw `tool/result` content. There is
* no call-time analogue: a search call stays a {@link GenericCallView}
* (`kind: 'search'`) because the pending state has no matches or paths to show —
* the structured shape exists only after `execute`.
*/
export type SearchResultView = SearchMatchesResultView | SearchPathsResultView
/**
* A completed file read rendered as a line-numbered, optionally syntax-highlighted
* code view by a capable UI. Set by a tool whose call reads file text (e.g.
* `read`); the pending state stays a {@link GenericCallView} (`kind: 'read'`)
* because a call carries no content until `execute` returns. The structured
* `lines`/`path`/`lang`/`totalLines` fields cannot be reconstructed from the
* model-facing result text alone, so the read tool projects them through its
* `output.presentationMeta` (persisted with the session log) and `presentResult`
* narrows that metadata back into this view on live and replay paths alike. A UI
* without the read capability falls back to `content` (the model-facing text with
* its envelope stripped), so this view degrades to the generic text card.
*/
export interface ReadResultView {
card: 'read'
/** Replacement title for the completed call. Omit to keep the pending-state title. */
title?: string
/** The read file's path (the model-facing path; the bridge relativizes it). */
path: string
/**
* The 1-based first line the window requested, preserved even when `lines` is
* empty (a byte cap below the first selected line yields an empty window) so a
* UI knows where the window starts and where a continuation resumes.
*/
offset: number
/** The returned window's lines, in file order, each keeping its file line number. */
lines: ReadFileLine[]
/** Exact total line count in the file, so a UI can show a "showing N of M" affordance. */
totalLines: number
/**
* A syntax-highlighting language hint derived from the file extension (e.g.
* `ts`, `py`), or omitted when the extension maps to no known language so a UI
* renders the lines as plain text.
*/
lang?: string
/**
* The model-facing result content with its envelope stripped, for a UI without
* the read capability. Omit to let such a UI render the raw result content.
*/
content?: ContentBlock[]
}
/**
* One citeable source in a completed {@link WebSearchResultView}, the faithful
* projection of one web-search source. The presentation projection of `dsh-web`'s