docs: rebalance prose cleanup and add trimming skill
This commit is contained in:
@@ -1,6 +1,8 @@
|
||||
/**
|
||||
* `LocalBashExecutor`: the local-subprocess implementation of the `@deepseek-ai/dsh-bash`
|
||||
* executor seam.
|
||||
* Local-subprocess implementation of the bash seam. Each call runs in its own
|
||||
* process group, background tasks are tracked, and disposal kills and awaits
|
||||
* them. Execution policy belongs in `tools/pre-execute` or a sandboxing
|
||||
* executor, not this local process layer.
|
||||
* @module @deepseek-ai/dsh-bash-local
|
||||
*/
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
/**
|
||||
* Process plumbing for the local bash executor: spawn, output collection with tail-keep +
|
||||
* spill-to-disk truncation, and process-group kill with SIGTERM→SIGKILL escalation.
|
||||
* Process plumbing for the local bash executor: detached process-group spawn,
|
||||
* tail-keep output with spill files, and SIGTERM→SIGKILL escalation. This layer
|
||||
* reacts to an abort signal; the executor owns deadlines and classifies causes.
|
||||
* @module dsh-bash-local/run
|
||||
*/
|
||||
|
||||
@@ -33,8 +34,8 @@ export const ENV_OVERRIDES = {
|
||||
export const SENSITIVE_ENV_PATTERN = /KEY|SECRET|TOKEN/i
|
||||
|
||||
/**
|
||||
* `process.env` minus credential-shaped vars, plus the model-friendly overrides, plus any
|
||||
* caller-supplied `extra` entries.
|
||||
* Build a child environment by scrubbing credential-shaped ambient variables,
|
||||
* applying model-friendly overrides, then merging trusted caller entries last.
|
||||
*
|
||||
* @param extra - caller-supplied entries merged last; an explicit entry wins even against the scrub and the overrides.
|
||||
* @returns the environment to hand to `spawn` for the child process.
|
||||
@@ -236,8 +237,8 @@ export class OutputCollector {
|
||||
try {
|
||||
closeSync(this.spillFd)
|
||||
} catch {
|
||||
// close can surface delayed writeback failures (for example EIO/ENOSPC) after writeSync
|
||||
// appeared to succeed.
|
||||
// A delayed writeback failure makes the spill unreliable; keep finalize
|
||||
// total but stop advertising that file.
|
||||
this.spillFile = undefined
|
||||
}
|
||||
this.spillFd = undefined
|
||||
@@ -247,9 +248,9 @@ export class OutputCollector {
|
||||
}
|
||||
|
||||
/**
|
||||
* Send `sig` to the process GROUP led by `pid` (requires the child to have been spawned with
|
||||
* `detached: true`).
|
||||
*
|
||||
* Send `sig` to a detached process group. Never throws: delivery races process
|
||||
* exit and may run in a timer callback, so failures are contained and a
|
||||
* non-positive pid is a no-op.
|
||||
* @param pid - the group leader's pid; non-positive means the spawn failed and the call is a no-op.
|
||||
* @param sig - the signal to deliver to the whole group.
|
||||
*/
|
||||
|
||||
@@ -337,7 +337,7 @@ describe('LocalBashExecutor background tasks', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('review fixes: lifecycle hardening', () => {
|
||||
describe('executor cancellation, callback, and disposal contracts', () => {
|
||||
it('start honors a pre-aborted or later-aborted AbortSignal', async () => {
|
||||
const { bash } = await setup()
|
||||
const controller = new AbortController()
|
||||
|
||||
@@ -189,8 +189,8 @@ describe('stdin and extra env (set by in-process plugins)', () => {
|
||||
})
|
||||
|
||||
it('gives fd 0 the exact pre-seam type: /dev/null when no stdin, a pipe when supplied', async () => {
|
||||
// The no-stdin path must stay observationally identical to the pre-seam `ignore` default: a
|
||||
// command that probes stdin's file type sees a char device (/dev/null).
|
||||
// With no bytes, fd 0 remains the pre-seam `ignore` default (/dev/null, a character device).
|
||||
// Supplied bytes use Node's spawn pipe, which is an AF_UNIX socket rather than a FIFO.
|
||||
const none = await runBash(spec('test -c /dev/stdin && echo char || echo other')).done
|
||||
expect(none.stdout.text).toBe('char\n')
|
||||
const piped = await runBash(spec('test -S /dev/stdin && echo socket || echo other', { stdin: 'x' })).done
|
||||
@@ -215,8 +215,8 @@ describe('stdin and extra env (set by in-process plugins)', () => {
|
||||
})
|
||||
|
||||
it('does not crash or reject when the child ignores a large stdin (EPIPE)', async () => {
|
||||
// The child exits immediately without reading; closing our end of a stdin pipe still
|
||||
// holding ~1MiB triggers EPIPE on the write.
|
||||
// The child exits without reading, so closing a stdin pipe holding ~1 MiB triggers EPIPE.
|
||||
// The handler swallows that write error and `done` reports the child's real exit.
|
||||
const big = 'x'.repeat(1024 * 1024)
|
||||
const result = await runBash(spec('exit 7', { stdin: big })).done
|
||||
expect(result.exitCode).toBe(7)
|
||||
@@ -355,7 +355,7 @@ describe('abort edge cases', () => {
|
||||
})
|
||||
})
|
||||
|
||||
describe('review fixes: env scrubbing and spill hardening', () => {
|
||||
describe('environment and spill-file hardening', () => {
|
||||
it('scrubs credential-shaped env vars from child processes', async () => {
|
||||
process.env.DSH_TEST_API_KEY = 'super-secret'
|
||||
process.env.DSH_TEST_TOKEN = 'also-secret'
|
||||
|
||||
@@ -14,7 +14,7 @@ Semantics:
|
||||
|
||||
- **Denials are result facts.** A failed run whose stderr carries the selected backend's own denial dialect — the signatures the provider stamps on every wrap (EROFS text under bwrap, EACCES under Landlock, EPERM under Seatbelt) — is reported as `BashRunResult.sandbox.denied: true` (conservative classification, read from the collected stderr tail); every CONFINED run also carries the mode it executed under (`result.sandbox.mode`) and the provider's enforcement completeness (`result.sandbox.enforcement`: `full`, or `partial` on an older Landlock ABI).
|
||||
- **Runner failures are sandbox failures, never task failures.** A failed run matching the wrap's `runnerFailureSignatures` (the runner's own error prefix — also what the shell prints for a missing runner) means the sandbox itself broke and the command NEVER RAN; the check outranks denial classification because a runner's error text can contain denial words. The foreground path re-throws it as the structured fail-closed `SANDBOX_UNAVAILABLE` error, with the runner's first stderr line as the cause; a settled background task stamps `task.sandbox.runnerFailed` instead (no error channel remains after settle), which `bash_output` renders as its own marker.
|
||||
- **Config default, per-call override.** `resolve()` stamps the configured sandbox mode onto each spec unless an approved request supplies a wider mode. That override affects only its call or background task. `ctx.bash.sandboxMode` reports the default so the tool advertises escalation only when supported; results report the effective mode.
|
||||
- **Config default, per-call override.** `resolve()` stamps the configured sandbox mode onto each spec unless an approved request supplies a wider mode. That override affects only its call or background task. `ctx.bash.sandboxMode` reports the default so the tool advertises escalation only when supported; results report the effective mode. The model learns standing mode only from tool/result facts, not a system-prompt announcement.
|
||||
- **File effects only.** Network and process visibility are deliberately not restricted — the mode vocabulary does not pretend to cover what the backend does not enforce.
|
||||
- Process mechanics (spawn, process-group kills, output collection/spill, background tasks, credential scrub) are inherited verbatim from [`dsh-bash-local`](../bash-local/); the runner ladder, probes, and the per-platform Landlock launcher packages live with [`dsh-sandbox-local`](../../sandbox/sandbox-local/).
|
||||
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
/**
|
||||
* `SandboxBashExecutor`: the sandbox-consuming implementation of the `@deepseek-ai/dsh-bash`
|
||||
* executor seam.
|
||||
* Sandbox-consuming bash executor. It wraps the exact local bash argv through
|
||||
* `ctx.sandbox`, inherits local process mechanics, and reports the selected
|
||||
* mode, enforcement, and denial facts. Runner failure means the command never
|
||||
* ran: foreground calls throw `SANDBOX_UNAVAILABLE`, while settled background
|
||||
* tasks carry `runnerFailed`. The tool owns approval and passes per-call modes.
|
||||
* @module @deepseek-ai/dsh-bash-sandbox
|
||||
*/
|
||||
|
||||
@@ -42,7 +45,9 @@ export function shellQuote(text: string): string {
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a nonzero run using the selected backend's denial signatures.
|
||||
* Conservatively classify a nonzero, non-signal run using only the selected
|
||||
* backend's denial signatures. Text inference may miss a denial or match
|
||||
* unrelated stderr in that dialect; it never uses another backend's terms.
|
||||
* @param result - the settled foreground run to classify.
|
||||
* @param signatures - the active wrap's denial dialect, case-insensitive stderr substrings.
|
||||
* @returns whether the run's failure reads as a sandbox denial.
|
||||
@@ -52,7 +57,9 @@ export function classifyDenial(result: BashRunResult, signatures: readonly strin
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a nonzero run using the selected backend's runner-failure signatures.
|
||||
* Classify a nonzero run using the selected backend's runner-failure
|
||||
* signatures. Callers check this before denial because runner diagnostics may
|
||||
* contain denial words; the command did not run.
|
||||
* @param result - the settled foreground run to classify.
|
||||
* @param signatures - the active wrap's runner-failure signatures,
|
||||
* case-insensitive stderr substrings.
|
||||
@@ -75,7 +82,9 @@ function matchesSignature(exitCode: number | null, stderr: string, signatures: r
|
||||
}
|
||||
|
||||
/**
|
||||
* Sandbox-consuming bash executor.
|
||||
* Registers as `ctx.bash` in place of the local executor and consumes a
|
||||
* `ctx.sandbox` provider. Its configured mode is the fallback; each resolved
|
||||
* call may carry a session override or approved one-shot escalation.
|
||||
*/
|
||||
export class SandboxBashExecutor extends LocalBashExecutor {
|
||||
static inject = ['sandbox']
|
||||
@@ -93,9 +102,9 @@ export class SandboxBashExecutor extends LocalBashExecutor {
|
||||
private readonly mode: SandboxMode
|
||||
private readonly workspaceRoot: string
|
||||
/**
|
||||
* Per-task facts, keyed by task id from `start()` until the settle stamp consumes them: the
|
||||
* mode the task runs under (per-call — an escalated task differs from its neighbors) plus
|
||||
* its wrap facts.
|
||||
* Per-task mode and wrap facts retained until settlement. Overlapping tasks
|
||||
* may use different modes or provider facts, so one latest-wrap field would
|
||||
* misclassify earlier completions.
|
||||
*/
|
||||
private readonly taskFacts = new Map<BashTaskId, {
|
||||
mode: ConfinedSandboxMode
|
||||
@@ -152,8 +161,8 @@ export class SandboxBashExecutor extends LocalBashExecutor {
|
||||
// Same stamped-by-resolve invariant as run().
|
||||
const mode = spec.sandboxMode as SandboxMode
|
||||
if (mode === 'danger-full-access') return super.start(spec)
|
||||
// Sandbox facts are stamped at settle time by {@link notifyTaskDone} (denial classification
|
||||
// runs against the settled task's collected stderr).
|
||||
// Classification needs settled stderr. Store facts synchronously after
|
||||
// spawn, before the earliest process completion can be observed.
|
||||
const confined = this.confine(spec.command, mode)
|
||||
const task = super.start({ ...spec, command: confined.command })
|
||||
const { enforcement, denialSignatures, runnerFailureSignatures } = confined
|
||||
@@ -162,17 +171,16 @@ export class SandboxBashExecutor extends LocalBashExecutor {
|
||||
}
|
||||
|
||||
/**
|
||||
* Stamp the sandbox facts before completion listeners run: the base executor notifies from
|
||||
* inside the task's settle path, so overriding the notification point is what makes
|
||||
* `task.sandbox` visible to `onTaskDone` consumers and `done` awaiters alike.
|
||||
* Stamp per-task sandbox facts before completion listeners and `done` settle.
|
||||
* Full-access tasks have no facts; signal deaths are not denials.
|
||||
*/
|
||||
protected override notifyTaskDone(task: BashTask): void {
|
||||
const facts = this.taskFacts.get(task.id)
|
||||
if (facts !== undefined) {
|
||||
this.taskFacts.delete(task.id)
|
||||
const stderr = this.collectedStderr(task.id)
|
||||
// Runner failure outranks denial (the command never ran; the runner's own error text can
|
||||
// contain denial words).
|
||||
// Runner failure outranks denial. Background settlement has no throw
|
||||
// channel, so this fact is its counterpart to the foreground exception.
|
||||
const runnerFailed = matchesSignature(task.exitCode, stderr, facts.runnerFailureSignatures)
|
||||
task.sandbox = {
|
||||
mode: facts.mode,
|
||||
|
||||
@@ -9,9 +9,13 @@ import { bwrapProfileArgs, LocalSandboxProvider } from '@deepseek-ai/dsh-sandbox
|
||||
import { SandboxBashExecutor } from '@deepseek-ai/dsh-bash-sandbox'
|
||||
|
||||
/**
|
||||
* Keyless consumer-integration proof under bwrap: the real `LocalSandboxProvider` (nothing
|
||||
* forced — bwrap is the ladder's first rung, so a passing probe selects it) underneath the
|
||||
* real `SandboxBashExecutor`, driven through the executor's public run/start paths.
|
||||
* Keyless integration of the real provider and executor through public run/start paths. With
|
||||
* no rung forced, a passing bwrap probe selects the ladder's first rung. The tests check world
|
||||
* effects and stamped facts, including EROFS classification through the wrap-carried dialect;
|
||||
* backend-only confinement is covered by `@deepseek-ai/dsh-sandbox-local`.
|
||||
*
|
||||
* Skips when bwrap or unprivileged user namespaces are unavailable. HOME-based paths are
|
||||
* intentional because bwrap replaces `/tmp`, which cannot prove the workspace-root boundary.
|
||||
*/
|
||||
|
||||
const probe = spawnSync('bwrap', [...bwrapProfileArgs({ mode: 'read-only', workspaceRoot: '/' }), '--', 'true'], { timeout: 5_000, stdio: 'ignore' })
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
/**
|
||||
* SandboxBashExecutor tests: the CONSUMER side of the sandbox seam.
|
||||
* Consumer-side `SandboxBashExecutor` tests. A fake Cordis sandbox service makes wrapping,
|
||||
* policy hand-off, fail-closed propagation, classification, and fact stamping deterministic;
|
||||
* real-provider integration lives in `tests/landlock.e2e.ts`. A mode-0555 directory supplies
|
||||
* the Unix denial signature used by the classifier without requiring a real sandbox runner.
|
||||
*/
|
||||
|
||||
import { chmodSync, mkdirSync, mkdtempSync } from 'node:fs'
|
||||
@@ -280,7 +283,9 @@ describe('background sandbox facts', () => {
|
||||
})
|
||||
|
||||
it('overlapping background tasks keep their OWN wrap facts (per-task, not latest-wrap)', async () => {
|
||||
// The seam returns facts per WRAP — a legal provider may vary them between calls.
|
||||
// Facts belong to each wrap and may vary between calls. The slow task settles after the
|
||||
// quick task starts; a shared latest-wrap field would classify and stamp it with the wrong
|
||||
// task's dialect and enforcement.
|
||||
const wraps: Array<Pick<ConfinedArgv, 'enforcement' | 'denialSignatures'>> = [
|
||||
{ enforcement: 'partial', denialSignatures: ['permission denied'] },
|
||||
{ enforcement: 'full', denialSignatures: ['read-only file system'] },
|
||||
|
||||
@@ -9,8 +9,11 @@ import { LocalSandboxProvider, seatbeltProfileArgs } from '@deepseek-ai/dsh-sand
|
||||
import { SandboxBashExecutor } from '@deepseek-ai/dsh-bash-sandbox'
|
||||
|
||||
/**
|
||||
* Keyless macOS integration of the real Seatbelt provider and sandbox executor,
|
||||
* including world effects and denial classification. Skips when the probe fails.
|
||||
* Keyless macOS integration of the real provider and executor through public run/start paths.
|
||||
* Linux rungs are forced off so Seatbelt is selected. The tests check world effects and stamped
|
||||
* facts, including EPERM classification through the wrap-carried dialect; backend-only
|
||||
* confinement is covered by `@deepseek-ai/dsh-sandbox-local`. Skips off macOS or when
|
||||
* `sandbox-exec` rejects the profile.
|
||||
*/
|
||||
|
||||
const probe = spawnSync('sandbox-exec', [...seatbeltProfileArgs({ mode: 'read-only', workspaceRoot: '/' }), '--', 'true'], { timeout: 5_000, stdio: 'ignore' })
|
||||
|
||||
@@ -29,9 +29,11 @@ declare module 'cordis' {
|
||||
}
|
||||
|
||||
/**
|
||||
* Abstract bash execution service. Subclass, implement the abstract methods, and load the
|
||||
* subclass as a plugin — it registers as `ctx.bash` (one implementation per context; loading a
|
||||
* second throws, which is cordis' standard duplicate-service behavior).
|
||||
* Registers one `ctx.bash` implementation. Runtime command failures resolve as
|
||||
* {@link BashRunResult}; only infrastructure failures reject. Background starts
|
||||
* return immediately without a timeout, report completion exactly once while
|
||||
* live, and remain cancellable by signal or {@link kill}. Output reads are
|
||||
* incremental and flag lost buffered data; disposal kills and awaits all tasks.
|
||||
*/
|
||||
export abstract class BashExecutor extends Service {
|
||||
private listeners = new Set<BashTaskListener>()
|
||||
@@ -51,7 +53,8 @@ export abstract class BashExecutor extends Service {
|
||||
* The sandbox mode this executor confines commands under BY DEFAULT, or `undefined` when it
|
||||
* does not sandbox at all — the capability fact the tool and ACP layers read to advertise
|
||||
* sandbox controls honestly.
|
||||
*
|
||||
* A session or call may override this default, so widening is evaluated per
|
||||
* execution rather than encoded in this getter.
|
||||
* @returns the configured default mode of a sandboxing executor;
|
||||
* `undefined` for an executor that never confines.
|
||||
*/
|
||||
@@ -92,7 +95,8 @@ export abstract class BashExecutor extends Service {
|
||||
/**
|
||||
* The opaque OWNER token recorded for a background task at {@link start} (from the {@link
|
||||
* BashExecSpec}'s `owner`), or `undefined` for an unknown id OR a known-but-ownerless task.
|
||||
*
|
||||
* The executor stores the token without interpreting policy; keeping it here
|
||||
* lets ownership survive a consumer-plugin reload.
|
||||
* @param id - the background task id to look up ownership for.
|
||||
* @returns the token recorded at start, verbatim; undefined for an unknown
|
||||
* id or a known-but-ownerless task.
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
/**
|
||||
* Per-session sandbox-mode override: the session log as the store.
|
||||
* Per-session sandbox-mode override stored as log-only events. Folding the log
|
||||
* isolates sessions and survives replay; the tool stamps the result onto each
|
||||
* call unless a one-shot escalation grant overrides it. The model sees the
|
||||
* effective mode through prompt guidance and boundary notices, not the event.
|
||||
* @module dsh-bash/session-mode
|
||||
*/
|
||||
|
||||
|
||||
@@ -119,8 +119,9 @@ export interface BashExecRequest {
|
||||
*/
|
||||
owner?: OwnerToken | undefined
|
||||
/**
|
||||
* Explicit per-call sandbox-policy input, overriding the executor's configured default mode
|
||||
* for this call.
|
||||
* Explicit per-call sandbox policy. The tool stamps a session override or a
|
||||
* one-shot approved escalation, with the grant taking precedence. Sandboxing
|
||||
* executors honor it for this call; non-sandboxing executors do not confine.
|
||||
*/
|
||||
sandboxMode?: SandboxMode | undefined
|
||||
}
|
||||
|
||||
@@ -4,7 +4,7 @@ The model-facing bash tools — `bash`, `bash_output`, `bash_kill` — registere
|
||||
|
||||
Requires a loaded executor implementation (e.g. `@deepseek-ai/dsh-bash-local`); the plugin stays pending until `ctx.bash` exists (`inject: ['tools', 'bash', 'systemPrompt']`).
|
||||
|
||||
The plugin also contributes the `tool:bash` prompt section (order 105) — the cross-call habit the per-tool descriptions cannot carry: check the `[exit code: N]` marker on every result and investigate failures before moving on. Under a sandboxing executor it additionally contributes the per-agent `env:bash-sandbox` section (order 110) stating each session's EFFECTIVE mode, and the pre-step narrator — see [Per-session mode](#per-session-mode-switching-and-visibility).
|
||||
The plugin also contributes the `tool:bash` prompt section (order 105) — the cross-call habit the per-tool descriptions cannot carry: check the `[exit code: N]` marker on every result and investigate failures before moving on. Sandbox mode is intentionally learned from denial results, not announced in the prompt; see [Per-session mode](#per-session-mode-switching).
|
||||
|
||||
## Tools
|
||||
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
/**
|
||||
* The model-facing bash tools: `bash`, `bash_output`, `bash_kill`. Pure schema + text shaping
|
||||
* — every process concern lives behind the `ctx.bash` executor seam (`@deepseek-ai/dsh-bash`),
|
||||
* so sandbox/permission/remote executor implementations swap in without touching what the
|
||||
* model sees.
|
||||
* Model-facing `bash`, `bash_output`, and `bash_kill` tools over the executor
|
||||
* seam. Background tasks are fenced by owning session, completion injects a
|
||||
* durable notice, and confining executors add one-shot approval-based escalation.
|
||||
* @module @deepseek-ai/dsh-tool-bash
|
||||
*/
|
||||
|
||||
@@ -25,7 +24,7 @@ export const name = 'tool-bash'
|
||||
export const inject = ['tools', 'bash', 'systemPrompt']
|
||||
|
||||
/**
|
||||
* Validate the constraints the SchemaSpec can't express. `defineTool` now
|
||||
* Validate the constraints the SchemaSpec can't express. `defineTool`
|
||||
* validates parsed args against the SchemaSpec before `execute` runs (the
|
||||
* arg-validation RFC), so type/required/enum checks are already done and `args`
|
||||
* is the validated `InferArgs` shape here. What remains are value constraints
|
||||
@@ -136,8 +135,9 @@ function streamText(output: CollectedOutput): string {
|
||||
}
|
||||
|
||||
/**
|
||||
* Shape one finished run into the text the model sees: stdout, then a marked stderr
|
||||
* section, then exit-status markers.
|
||||
* Shape one finished run into model-visible stdout, marked stderr, and status
|
||||
* facts. Non-zero exits and sandbox denials remain ordinary results; only
|
||||
* infrastructure failure or abort makes the tool call itself fail.
|
||||
*
|
||||
* @param result - the completed foreground run from the executor.
|
||||
* @param escalationModes - the escalation targets this composition advertises; non-empty
|
||||
@@ -193,7 +193,7 @@ export function renderResult(
|
||||
// UI presentation (tool-owned).
|
||||
|
||||
/**
|
||||
* Pending-state presentation for a `bash` call.
|
||||
* Present foreground calls as terminals and background starts as generic cards.
|
||||
*/
|
||||
type BashCallArgs = { command: string; description: string; workdir?: string; run_in_background?: boolean }
|
||||
|
||||
@@ -220,7 +220,8 @@ function presentBashCall(args: BashCallArgs): GenericCallView | TerminalCallView
|
||||
}
|
||||
|
||||
/**
|
||||
* Completed-state presentation for a `bash` call.
|
||||
* Present completed foreground output as a terminal; background acknowledgements
|
||||
* and execution errors use generic fenced output without an exit-status pill.
|
||||
*/
|
||||
function presentBashResult(args: unknown, result: ToolResult): ToolResultView | undefined {
|
||||
const block = result.content.length === 1 ? result.content[0] : undefined
|
||||
@@ -238,8 +239,8 @@ function presentBashResult(args: unknown, result: ToolResult): ToolResultView |
|
||||
}
|
||||
|
||||
/**
|
||||
* Recover the structured exit status from a rendered `renderResult` string — the inverse of
|
||||
* the status markers it appends.
|
||||
* Recover exit status from the final marked line emitted by {@link renderResult}.
|
||||
* A program whose own final line exactly mimics a marker remains ambiguous for UI display.
|
||||
*/
|
||||
function parseExitStatus(text: string): { exitCode: number } | { signal: string } {
|
||||
const signal = /\n\[killed by signal: ([^\]\n]+)\]$/.exec(text)
|
||||
@@ -255,7 +256,8 @@ function presentTaskCall(verb: string, args: { task_id: string }): GenericCallVi
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the working directory for a bash call.
|
||||
* Resolve an explicit workdir first, making a relative one session-cwd-relative;
|
||||
* otherwise use the session cwd and leave executor defaulting as the fallback.
|
||||
*/
|
||||
function resolveWorkdir(modelWorkdir: string | undefined, exec: { agent?: Agent }): string | undefined {
|
||||
const sessionCwd = exec.agent?.session.header.cwd
|
||||
@@ -314,7 +316,8 @@ export function apply(ctx: Context): void {
|
||||
}
|
||||
}
|
||||
|
||||
// Background completion → inject a notice into the owning agent's session.
|
||||
// Completion runs on the bash fiber, so use topology-independent lookup and
|
||||
// match the executor's stored session-owner token to a live agent.
|
||||
ctx.bash.onTaskDone((task) => {
|
||||
const ownerToken = ctx.bash.ownerOf(task.id)
|
||||
if (ownerToken === undefined) return
|
||||
|
||||
@@ -56,7 +56,8 @@ async function setup() {
|
||||
*/
|
||||
const fakeAgentDisposers = new Map<Context, (() => Promise<void> | void)[]>()
|
||||
function registerFakeAgent(ctx: Context, sessionId: string, inject: (...args: unknown[]) => void): Agent {
|
||||
// Distinct ids ensure notices match the session owner token, not the registry key.
|
||||
// A config agent has distinct registry (`agent.id`) and owner (`session.header.id`) tokens.
|
||||
// Keeping them unequal makes notice lookup by the wrong field fail instead of passing by chance.
|
||||
const agent = { id: `agent-${sessionId}`, inject, session: { header: { version: 0, id: sessionId, createdAt: 0 } } } as unknown as Agent
|
||||
const dispose = ctx.agents.register(agent)
|
||||
const list = fakeAgentDisposers.get(ctx) ?? []
|
||||
@@ -411,8 +412,8 @@ describe('background tools', () => {
|
||||
it('injects a completion notice into the owning agent (found via the registry by session token)', async () => {
|
||||
const ctx = await setup()
|
||||
const inject = vi.fn()
|
||||
// The notice path looks the agent up in ctx.agents by its session token, so the agent must
|
||||
// be REGISTERED (not merely passed to execute).
|
||||
// Notices look up the agent in ctx.agents by session token, so passing it to execute is not
|
||||
// enough: the fake must be registered with a matching `session.header.id`.
|
||||
const agent = registerFakeAgent(ctx, 'bg', inject)
|
||||
|
||||
const started = await ctx.tools.execute({
|
||||
@@ -475,9 +476,8 @@ describe('background tools', () => {
|
||||
})
|
||||
|
||||
it('drops the notice cleanly when the owning agent is gone from the registry by completion', async () => {
|
||||
// A bash task (owned by the host-scoped bash-local fiber) can OUTLIVE its per-session agent
|
||||
// — e.g. the ACP session disconnects and its AgentHandle disposes while the background task
|
||||
// is still running.
|
||||
// Host-scoped bash tasks can outlive a per-session agent after an ACP disconnect. The task
|
||||
// retains its owner token, but with no matching live agent the notice is dropped without error.
|
||||
const ctx = await setup()
|
||||
const inject = vi.fn()
|
||||
const agent = registerFakeAgent(ctx, 'bg', inject)
|
||||
@@ -507,9 +507,8 @@ describe('background task ownership (cross-session isolation)', () => {
|
||||
function callAs(ctx: Context, agent: import('@deepseek-ai/dsh-agent').Agent | undefined, name: string, args: unknown) {
|
||||
return ctx.tools.execute({ callId: CallId(`own-${++callCounter}`), name, arguments: args, ...agent ? { agent } : {} })
|
||||
}
|
||||
// Ownership is by TOKEN (session.header.id), not agent object identity — so each agent needs
|
||||
// a DISTINCT session id, else every fake yields the same token and the isolation tests pass
|
||||
// for the wrong reason (all tasks owned by the same token).
|
||||
// Ownership uses `session.header.id`, not object identity. Distinct ids keep the isolation tests
|
||||
// from passing accidentally because every fake produced the same owner token.
|
||||
const fakeAgent = (sessionId: string) =>
|
||||
({ inject: () => undefined, session: { header: { version: 0, id: sessionId, createdAt: 0 } } }) as unknown as import('@deepseek-ai/dsh-agent').Agent
|
||||
|
||||
@@ -589,8 +588,8 @@ describe('background task ownership (cross-session isolation)', () => {
|
||||
})
|
||||
|
||||
it('ownership SURVIVES an independent tool-bash HMR reload (token lives on the executor)', async () => {
|
||||
// The owner token lives on the TASK inside the executor (dsh-bash fiber), not in a
|
||||
// tool-bash plugin-local map.
|
||||
// The executor task owns the token, so reloading only tool-bash preserves ownership. A
|
||||
// plugin-local map would lose it and incorrectly expose the task to agent B.
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRegistry)
|
||||
@@ -809,9 +808,8 @@ describe('tool-owned UI presentation (presentCall / presentResult)', () => {
|
||||
it('bash presentResult: a clean exit-0 whose output ENDS in marker-like text is NOT read as a failure', async () => {
|
||||
const ctx = await setup()
|
||||
const args = { command: 'printf "[exit code: 5]"', description: 'print' }
|
||||
// A successful command can print text that looks like a marker. renderResult for a clean
|
||||
// exit 0 appends NOTHING (and no trailing newline), so the body's own tail is `[exit code:
|
||||
// 5]`.
|
||||
// A successful command may print marker-like text. A clean result appends no marker or
|
||||
// newline; parsing requires the leading newline emitted for real markers, so this stays exit 0.
|
||||
const out = ctx.tools.get('bash')!.presentResult!(args, { content: [{ type: 'text', text: '[exit code: 5]' }], isError: false })
|
||||
expect(out).toEqual({ card: 'terminal', output: '[exit code: 5]', exitCode: 0 })
|
||||
// Same for a fake signal marker with no leading newline.
|
||||
@@ -874,17 +872,18 @@ describe('tool-owned UI presentation (presentCall / presentResult)', () => {
|
||||
|
||||
it('presentCall validates softly: malformed args (missing required description) return undefined, never throw', async () => {
|
||||
const ctx = await setup()
|
||||
// defineTool wraps presentCall to soft-validate against the schema and fall back to
|
||||
// undefined (a generic UI presentation) rather than throwing on the display path — it may
|
||||
// run on replay of arbitrary logged args.
|
||||
// `defineTool` soft-validates replayed logged args before presentation. Invalid shapes return
|
||||
// undefined for generic UI rendering rather than throwing; `presentCall` accepts `unknown`.
|
||||
expect(ctx.tools.get('bash')?.presentCall?.({ command: 'ls' })).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('the model-facing bash tool builds its request from named args only (no {...args} forward)', () => {
|
||||
/**
|
||||
* Records every {@link BashExecRequest} the consumer hands to `resolve()`, so a test can
|
||||
* assert what the model-facing tool DID and DID NOT forward.
|
||||
* Records requests passed to `resolve()` so tests can prove the model-facing tool forwards only
|
||||
* named arguments. It intentionally exposes neither `stdin` nor `env`; this catches a future
|
||||
* `...args` spread into the post-scrub env merge. The credential scrub remains the security
|
||||
* boundary; see the bash stdin/env RFC. Foreground `run()` is canned and `start()` is unused.
|
||||
*/
|
||||
class RecordingBashExecutor extends BashExecutor {
|
||||
readonly requests: BashExecRequest[] = []
|
||||
@@ -927,7 +926,9 @@ describe('the model-facing bash tool builds its request from named args only (no
|
||||
|
||||
it('does not forward env/stdin even when the model includes them as extra arguments', async () => {
|
||||
const { ctx, bash } = await setupRecording()
|
||||
// Extra args: the model includes `env` and `stdin` keys hoping they reach the executor.
|
||||
// Unknown `env` and `stdin` keys are ignored by the schema and named request construction.
|
||||
// This preserves the request shape; it is not a security boundary because shell syntax can
|
||||
// already set environment variables or feed stdin.
|
||||
await ctx.tools.execute({
|
||||
callId: CallId('no-forward-1'),
|
||||
name: 'bash',
|
||||
@@ -1418,8 +1419,9 @@ describe('per-session sandbox mode (the bash/sandbox-mode fold)', () => {
|
||||
})
|
||||
|
||||
it('escalates relative to the session effective mode, not the executor default (narrower override)', async () => {
|
||||
// The blocker scenario: a workspace-write default with a read-only override — the sensible
|
||||
// escalation is workspace-write, which a default-relative ladder could not even express.
|
||||
// With a workspace-write default and read-only override, escalation must return to
|
||||
// workspace-write. The static target vocabulary exposes it, and validation compares it with
|
||||
// the call's effective override rather than a default-relative ladder.
|
||||
const ctx = await setupModal('workspace-write', { approval: true })
|
||||
ctx.on('approval/request', () => Promise.resolve<ApprovalOutcome>('allowed-once'))
|
||||
const seen: (string | undefined)[] = []
|
||||
|
||||
Reference in New Issue
Block a user