feat: return typed values from Code Mode
This commit is contained in:
@@ -28,8 +28,8 @@ Use the ralph tool ONLY when the direct human explicitly asks for a Ralph loop o
|
||||
|
||||
Pass `run_code` the body of an async TypeScript function (erasable syntax only — no `enum` or namespaces; type annotations are advisory, the code runs type-stripped). Inside the program:
|
||||
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's text output as a string. Tool arguments must be JSON-serializable.
|
||||
- A FAILED tool call rejects with an `Error` carrying the tool's error text — `try/catch` it to handle and continue.
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's typed canonical JSON value. Tool arguments must be lossless JSON.
|
||||
- A FAILED tool call rejects with `ToolCallError`, whose `toolName` identifies the failed tool and whose `message` is human-readable — `try/catch` it to handle and continue.
|
||||
- Calls execute sequentially, even under `Promise.all`.
|
||||
- Emit results with `return` and/or `console.log(...)`. ONLY what you print or return comes back to you — intermediate tool results never enter the conversation, so extract just what you need.
|
||||
|
||||
@@ -38,9 +38,9 @@ The available tools:
|
||||
```ts
|
||||
type JsonValue = null | boolean | number | string | JsonValue[] | { [key: string]: JsonValue }
|
||||
|
||||
declare const tools: {
|
||||
interface ToolArgsMap {
|
||||
/** Execute a bash command (`bash -c`) and return its stdout/stderr. Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. Current harness environment facts are exposed through managed `$DSH_*` variables; inspect them when needed. Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way. Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. Set `run_in_background: true` for long-running commands: the call returns a task id immediately; read its output with `task_output` and stop it with `task_kill`. Attempting a command the sandbox may deny is safe and expected: run it and read the marker rather than assuming the denial. When a command is denied and a wider mode would let it succeed, escalate immediately in the same turn — the one sanctioned exception to a denial: retry the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) plus a one-sentence `justification`. Do not detour through chat to ask permission first — the approval prompt raised by that retry is how the user consents. If the session states approval prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. Never escalate speculatively: ground the request in a real denial — normally the one this command just hit; escalating up front is fine only when this session already denied the same access. A rejected escalation is final for that command — stop and explain, never work around it — but it does not forbid attempting or escalating other commands later. */
|
||||
bash(args: {
|
||||
bash: {
|
||||
/** The bash command to execute. */
|
||||
command: string;
|
||||
/** Clear, concise description of what this command does in active voice, 5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; "git status" → "Show working tree status"; "npm install" → "Install package dependencies". */
|
||||
@@ -55,33 +55,33 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact command needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Inspect the live cordis runtime that is running THIS agent. Read-only. Sections: `services` (every provided ctx service and the plugin fiber that owns it), `plugins` (a flat list of the loaded plugins with their lifecycle states), `tools` (the model-facing tools currently registered, i.e. what you can call), `dynamic` (plugins you mounted via cordis_mount: id, name, state, provided services, awaited services), `api` (method signatures AND argument/return type shapes for every LIVE service — read this before writing plugin code that calls a service), `events` (every harness event with its dispatch mode and exact signature — pick listener targets here). Omit `what` to get all six sections. With `what:"api"` or `what:"events"`, pass an exact `name` to narrow to one service/event and include its original source JSDoc. */
|
||||
cordis_inspect(args: {
|
||||
cordis_inspect: {
|
||||
/** Limit the report to one section. Omit for all sections. */
|
||||
what?: "services" | "plugins" | "tools" | "dynamic" | "api" | "events";
|
||||
/** Exact service key or event name whose original JSDoc to include; valid only with what:"api" or what:"events". */
|
||||
name?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Mount a NEW cordis plugin into the live runtime that is running THIS agent (self-modification). `code` runs as the body of an async JavaScript function in an isolated sandbox and MUST `return` a plugin. Two forms: FUNCTION form `return (ctx) => { … }` — declares no inject, so it can register tools, listen to events, and provide services, but reaching ANY service (e.g. ctx.bash) throws; use it only when you need no services. OBJECT form `return { name?, inject: ['bash', 'llm', …], apply(ctx) { … } }` — declares dependencies, and cordis activates the plugin only after the services exist; PREFER this form. You may reach ONLY the services you list in inject: an undeclared service throws even if it exists, because an undeclared dependency would not be cleaned up if its provider is unmounted. BEFORE calling a service from your code, read cordis_inspect what:"api" — it lists method signatures AND the type shapes of their arguments/returns (do not guess a field's type; e.g. a bash run's stdout is an object, not a string). Inside `apply`, use the standard cordis API: `ctx.on(event, listener)` to observe events (see cordis_inspect what:"events"), or call `harness.registerTool(ctx, harness.defineTool({ name, description, parameters: { text: { type: 'string', required: true } }, output: { schema: { type: 'string' }, render(_args, value) { return [{ type: 'text', text: value }] } }, async execute(args) { return args.text } }))` to give yourself a new tool — it becomes callable on your NEXT step. Tool parameters: each key IS a property — { type: 'string'|'number'|'integer'|'boolean'|'null'|'object'|'array'|'json', required?: true, description?, enum?, const?, items?, properties? }; every direct DSL object declares additionalProperties: true|false, and oneOf: [schema, schema, ...] replaces type for an exact-one union. A raw JSON-Schema { type: 'object', properties, required?: […] } wrapper is also accepted with open-by-default objects. A tool's `execute` MUST return the lossless JSON value declared by `output.schema`; `output.render(args, value)` separately returns Native/model content blocks. Mounts can COMPOSE: one plugin may `ctx.provide('name', value)` a service and another may declare `inject: ['name']` to consume it — the consumer stays pending until the provider exists and returns to pending when the provider is unmounted. Everything registered inside `apply` is cleaned up automatically on unmount. Sandbox globals: `console` (tagged `[cordis:<id>]`, writes through to the harness terminal), `harness.defineTool`, `harness.registerTool`, `btoa`, `atob`, `TextEncoder`, `TextDecoder`. Node APIs are DISABLED — do filesystem/network/timer work through the cordis services, never Node built-ins: `require`, `setTimeout`/`setInterval`, and `fetch` throw redirect errors; `process` and `Buffer` are undefined. Instead use inject: ['fs'] + ctx.fs for files, inject: ['web'] + ctx.web for HTTP, inject: ['bash'] + ctx.bash for processes, and inject: ['timer'] + ctx.setTimeout/ctx.setInterval for timing (fiber effects, auto-cleaned on unmount) — cordis_inspect what:"api" shows what THIS runtime provides. Write PLAIN JavaScript, not TypeScript (no `as`, no type annotations). Cautions: (1) waterfall events (e.g. tools/pre-execute) hand the listener a trailing `next` callback which MUST be called — returning without `next()` VETOES the call; prefer plain notification events unless you intend to intercept. (2) Never await something that only resolves after the current turn (your code runs INSIDE a tool call of that turn — it would deadlock). (3) Your `ctx` is a restricted façade: you can register tools, observe events, provide/consume services, and use timers, but framework internals (ctx.root, ctx.fiber, ctx.extend, ctx.plugin, …) are withheld. It is not a security boundary though — the services you inject (e.g. ctx.bash) reach the real runtime. */
|
||||
cordis_mount(args: {
|
||||
cordis_mount: {
|
||||
/** Body of an async JS function; must `return` the plugin to mount. */
|
||||
code: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Dispose a plugin previously mounted with cordis_mount, by id. All its registrations (event listeners, tools, services) are cleaned up through the cordis effect lifecycle. Returns only after disposal has fully completed (quiescence, not just a request to stop). */
|
||||
cordis_unmount(args: {
|
||||
cordis_unmount: {
|
||||
/** The dynamic mount id returned by cordis_mount (e.g. "dyn-1"). */
|
||||
id: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create one persisted same-session completion goal when the current direct human request is a long-running objective that should continue across autonomous goal rounds. You may infer that intent without requiring the user to say "create a goal". Do not use this for trivial single-turn work. Execution rejects non-human and subagent authority. */
|
||||
create_goal(args: {
|
||||
create_goal: {
|
||||
/** The concrete completion objective inferred from the direct human request. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer limit on automatic continuation rounds. */
|
||||
max_goal_rounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Edit an existing UTF-8 text file by replacing literal text. */
|
||||
edit(args: {
|
||||
edit: {
|
||||
/** Path to edit, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Literal text to replace. Must match exactly. */
|
||||
@@ -94,68 +94,68 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read the current same-session goal, including its exact id/revision, objective, phase, completed continuation rounds, round limit, blocker reason when present, and whether another continuation is armed. Call this before updating a goal. */
|
||||
get_goal(args: Record<string, JsonValue>): Promise<string>;
|
||||
get_goal: Record<string, JsonValue>;
|
||||
/** Run a foreground fresh-agent Ralph loop toward one immutable objective. Use only when the direct human explicitly asks for Ralph or fresh-agent iteration. Each round opens a new child with no parent conversation or prior child session; the shared workspace is long-term memory, and only a bounded structured report crosses rounds. The call returns when a worker reports completion or a concrete blocker, or at the round limit. Ordinary long-running same-session work belongs to goal tools. */
|
||||
ralph(args: {
|
||||
ralph: {
|
||||
/** The immutable completion objective for every fresh Ralph round. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer round cap, bounded by the deployment ceiling. */
|
||||
maxRounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read a UTF-8 text file and return line-numbered content. */
|
||||
read(args: {
|
||||
read: {
|
||||
/** Path to read, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** 1-based first line to return. Defaults to 1. */
|
||||
offset?: number;
|
||||
/** Maximum number of lines to return. Defaults to 2000. */
|
||||
limit?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Load the full instructions for an available skill. Call this with the exact skill name from the session skill catalog before acting on a task that names or clearly matches that skill. */
|
||||
skill(args: {
|
||||
skill: {
|
||||
/** The exact skill name from the available skills list. */
|
||||
name: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a self-contained task to a subagent (a separate agent that works in its own context) and return its final result. Use this to offload focused, independent work — research, a scoped implementation, an analysis — so it does not consume this conversation's context. The subagent runs to completion and you receive only its final answer, not its intermediate steps. Give it a complete, standalone prompt: it does not see this conversation. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent(args: {
|
||||
subagent: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The complete, self-contained task for the subagent. It does not share this conversation's context, so include everything it needs. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a task to a subagent that inherits this conversation: a child agent seeded with all completed turns so far (it does not see the current in-flight turn), returning only its final result. Use this when the subtask builds on this conversation's context — a follow-up analysis, a review, a continuation — without consuming this conversation's context for the work itself. You receive only its final answer, not its intermediate steps. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent_fork(args: {
|
||||
subagent_fork: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The task for the subagent. It already sees this conversation's completed turns, so build on them freely and state only what is new. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Request cancellation of a running background task by task id. Returns immediately; the task settles as killed once its work actually stops. */
|
||||
task_kill(args: {
|
||||
task_kill: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Optional short reason, recorded in the log and forwarded to the task. */
|
||||
reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** List your background tasks (running and finished) with their ids, kinds, and statuses. */
|
||||
task_list(args: Record<string, JsonValue>): Promise<string>;
|
||||
task_list: Record<string, JsonValue>;
|
||||
/** Read a background task. Stream tasks return only output since the previous read; final-output tasks return their result after settlement. Every response ends with `[status: ...]`. Reads are non-blocking unless `wait: true`, which waits up to the configured cap. */
|
||||
task_output(args: {
|
||||
task_output: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Block until the task reaches a terminal status or the timeout expires. A timed-out wait returns [status: running] and leaves the task alive. */
|
||||
wait?: boolean;
|
||||
/** Max wait in milliseconds (only meaningful with wait: true). Defaults to the configured wait timeout; capped by the configured maximum. */
|
||||
timeout_ms?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Record and update a structured task list for the current work. Send the ENTIRE list every call — it REPLACES the previous list (there are no partial updates, no per-item edits). Use it to plan multi-step work and show progress: add one todo per concrete step before you start. Keep AT MOST ONE todo `in_progress` at a time; while work remains, exactly one active task should be `in_progress`. Mark a todo `completed` the moment it is done (do not batch completions), and allow no `in_progress` item only once all work is complete. Skip the list for trivial single-step tasks. Statuses: `pending` (not started), `in_progress` (being worked on now), `completed` (finished). */
|
||||
todo_write(args: {
|
||||
todo_write: {
|
||||
/** The COMPLETE task list, replacing any previous list. */
|
||||
todos: ({
|
||||
/** What the task is — a short imperative line. */
|
||||
@@ -163,9 +163,9 @@ declare const tools: {
|
||||
/** pending (not started) | in_progress (now) | completed (done). */
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
} & Record<string, JsonValue>)[];
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Update the exact current goal revision. edit, pause, and resume require a direct top-level human request. During an automatic continuation of the current goal, complete and blocked are also allowed. blocked is rejected before the configured minimum round count; the model remains responsible for judging that the same condition persisted across those rounds and must explain it in blocked_reason. */
|
||||
update_goal(args: {
|
||||
update_goal: {
|
||||
/** Exact id returned by get_goal. */
|
||||
goal_id: string;
|
||||
/** Exact positive revision returned by get_goal. */
|
||||
@@ -178,9 +178,9 @@ declare const tools: {
|
||||
max_goal_rounds?: number;
|
||||
/** Concrete blocking condition; required only with action blocked. */
|
||||
blocked_reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Run a JavaScript workflow script that orchestrates subagents at scale. Use this for work that fans out across many independent pieces — an audit over many files, a migration, multi-angle research, adversarial verification of findings — where you write the orchestration as a script instead of delegating turn by turn. The workflow's identity rides the `meta` parameter as JSON: required `name` (short kebab-case) and `description` strings, optional `whenToUse` string and `phases` array (`{title, detail?, provider?, model?}`). The `script` parameter is the plain JavaScript body ONLY (NOT TypeScript, and NO `export const meta` statement — meta is a parameter, not code), running with top-level await; end with `return <value>` — the value must be JSON-serializable and is this tool's result. Script-body hooks: - `agent(prompt, opts?): Promise<any>` — run one subagent to completion. Without `opts.schema` it resolves to the child's final text; with `opts.schema` (an object-rooted JSON Schema using ONLY type/properties/required/additionalProperties/items/enum/const — no oneOf/pattern/format/numeric bounds) it resolves to the validated object. Resolves `null` when the child fails (filter with `.filter(Boolean)`). Other opts: `label` (display), `phase` (progress group), and independent `provider`/`model` LLM target overrides (either may be provided alone). Anything else (`effort`/`isolation`/`agentType`) is rejected loudly. - `pipeline(items, ...stages): Promise<any[]>` — run each item through the stages independently with NO barrier between stages (prefer this for multi-stage work). Each stage receives `(prev, item, index)`. An ordinary stage throw drops that ITEM to `null` and skips its remaining stages. - `parallel(thunks): Promise<any[]>` — run zero-argument functions concurrently and await ALL of them (a barrier; use only when a stage genuinely needs every prior result together). A throwing thunk resolves to `null`. - `phase(title)` — start a progress phase; `log(message)` — narrate progress; `args` — the tool call's `args` input, verbatim. Misused hooks (bad arguments, unknown options, unsupported schemas, tripped caps) throw errors that ALWAYS kill the script — they never dissolve into a per-item `null`. Constraints: concurrency and total-agent caps apply; no filesystem, network, timers, or Node.js APIs are provided — the agents do the work, the script only coordinates them. The run executes in the foreground: this call returns when the whole script finishes. */
|
||||
workflow(args: {
|
||||
workflow: {
|
||||
/** The plain-JS workflow script body (top-level await allowed; NO `export const meta` statement; end with `return <json-value>`). */
|
||||
script: string;
|
||||
/** The workflow identity block (plain JSON — never code). */
|
||||
@@ -205,9 +205,9 @@ declare const tools: {
|
||||
} & Record<string, JsonValue>;
|
||||
/** Optional JSON input exposed to the script as the `args` global (wrap a bare list as a field, e.g. {"files": [...]}). */
|
||||
args?: Record<string, JsonValue>;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create or fully replace a UTF-8 text file. */
|
||||
write(args: {
|
||||
write: {
|
||||
/** Path to write, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Full UTF-8 text content to write. */
|
||||
@@ -216,6 +216,215 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
}
|
||||
|
||||
interface ToolOutputMap {
|
||||
bash: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
exitCode: number | null;
|
||||
signal: string | null;
|
||||
timedOut: boolean;
|
||||
aborted: boolean;
|
||||
timeoutMs: number;
|
||||
stdout: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
stderr: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
sandbox?: {
|
||||
mode: string;
|
||||
denied: boolean;
|
||||
enforcement?: string;
|
||||
runnerFailed?: boolean;
|
||||
};
|
||||
};
|
||||
cordis_inspect: string;
|
||||
cordis_mount: {
|
||||
id: string;
|
||||
pluginName: string;
|
||||
state: "pending" | "loading" | "active" | "failed" | "disposed" | "unloading";
|
||||
provides: string[];
|
||||
waitingFor: string[];
|
||||
};
|
||||
cordis_unmount: {
|
||||
id: string;
|
||||
pluginName: string;
|
||||
};
|
||||
create_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
edit: {
|
||||
path: string;
|
||||
before: string;
|
||||
after: string;
|
||||
};
|
||||
get_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
ralph: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
read: {
|
||||
path: string;
|
||||
offset: number;
|
||||
lines: {
|
||||
number: number;
|
||||
text: string;
|
||||
}[];
|
||||
totalLines: number;
|
||||
};
|
||||
skill: {
|
||||
name: string;
|
||||
provider: string;
|
||||
resourceBase?: {
|
||||
kind: "directory";
|
||||
path: string;
|
||||
} | {
|
||||
kind: "url";
|
||||
url: string;
|
||||
} | {
|
||||
kind: "opaque";
|
||||
description: string;
|
||||
};
|
||||
content: string;
|
||||
};
|
||||
subagent: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
subagent_fork: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
task_kill: {
|
||||
outcome: "cancellation-requested" | "already-finished";
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
task_list: ({
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
})[];
|
||||
task_output: {
|
||||
text: string;
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
todo_write: {
|
||||
todos: ({
|
||||
content: string;
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
})[];
|
||||
counts: {
|
||||
pending: number;
|
||||
inProgress: number;
|
||||
completed: number;
|
||||
};
|
||||
};
|
||||
update_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
workflow: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
write: {
|
||||
path: string;
|
||||
operation: "create" | "update";
|
||||
before: string | null;
|
||||
after: string;
|
||||
};
|
||||
}
|
||||
|
||||
type ToolName = keyof ToolOutputMap
|
||||
|
||||
declare class ToolCallError extends Error {
|
||||
readonly name: "ToolCallError";
|
||||
readonly toolName: ToolName;
|
||||
}
|
||||
|
||||
declare const tools: {
|
||||
[K in ToolName]: (args: ToolArgsMap[K]) => Promise<ToolOutputMap[K]>;
|
||||
}
|
||||
```
|
||||
|
||||
@@ -76,16 +76,16 @@
|
||||
{"type":"assistant/chunk","seq":74,"time":1783611775407,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":"\\\""}}}
|
||||
{"type":"assistant/chunk","seq":75,"time":1783611775408,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":" });\\n"}}}
|
||||
{"type":"assistant/chunk","seq":76,"time":1783611775408,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":"return"}}}
|
||||
{"type":"assistant/chunk","seq":77,"time":1783611775436,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":" result"}}}
|
||||
{"type":"assistant/chunk","seq":77,"time":1783611775436,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":" result.stdout.text"}}}
|
||||
{"type":"assistant/chunk","seq":78,"time":1783611775437,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":";"}}}
|
||||
{"type":"assistant/chunk","seq":79,"time":1783611775437,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":"\""}}}
|
||||
{"type":"assistant/chunk","seq":80,"time":1783611775474,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","argumentsDelta":"}"}}}
|
||||
{"type":"assistant/chunk","seq":81,"time":1783611775497,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"reasoning","text":"The user wants me to use the `run_code` tool to execute a program that calls `tools.bash` with the command `echo BOTH_OK` and returns its output."}}}}
|
||||
{"type":"assistant/chunk","seq":82,"time":1783611775498,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","arguments":"{\"code\": \"const result = await tools.bash({ command: \\\"echo BOTH_OK\\\", description: \\\"Echo BOTH_OK\\\" });\\nreturn result;\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":82,"time":1783611775498,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","arguments":"{\"code\": \"const result = await tools.bash({ command: \\\"echo BOTH_OK\\\", description: \\\"Echo BOTH_OK\\\" });\\nreturn result.stdout.text;\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":83,"time":1783611775498,"data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":5530,"outputTokens":108,"cacheReadTokens":0,"reasoningTokens":37}}}}
|
||||
{"type":"assistant/chunk","seq":84,"time":1783611775498,"data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"tool-calls"}}}}
|
||||
{"type":"assistant/message","seq":85,"time":1783611775503,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants me to use the `run_code` tool to execute a program that calls `tools.bash` with the command `echo BOTH_OK` and returns its output."},{"type":"tool-call","id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","arguments":"{\"code\": \"const result = await tools.bash({ command: \\\"echo BOTH_OK\\\", description: \\\"Echo BOTH_OK\\\" });\\nreturn result;\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":5530,"outputTokens":108,"cacheReadTokens":0,"reasoningTokens":37}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":86,"time":1783611775504,"data":{"turn":1,"step":1,"callId":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","arguments":"{\"code\": \"const result = await tools.bash({ command: \\\"echo BOTH_OK\\\", description: \\\"Echo BOTH_OK\\\" });\\nreturn result;\"}"}}
|
||||
{"type":"assistant/message","seq":85,"time":1783611775503,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants me to use the `run_code` tool to execute a program that calls `tools.bash` with the command `echo BOTH_OK` and returns its output."},{"type":"tool-call","id":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","arguments":"{\"code\": \"const result = await tools.bash({ command: \\\"echo BOTH_OK\\\", description: \\\"Echo BOTH_OK\\\" });\\nreturn result.stdout.text;\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":5530,"outputTokens":108,"cacheReadTokens":0,"reasoningTokens":37}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":86,"time":1783611775504,"data":{"turn":1,"step":1,"callId":"call_00_AZFzvUwuC4vAUoICrfke5147","name":"run_code","arguments":"{\"code\": \"const result = await tools.bash({ command: \\\"echo BOTH_OK\\\", description: \\\"Echo BOTH_OK\\\" });\\nreturn result.stdout.text;\"}"}}
|
||||
{"type":"tool/code-dispatch","seq":87,"time":1783611775590,"data":{"parentCallId":"call_00_AZFzvUwuC4vAUoICrfke5147","subCallId":"call_00_AZFzvUwuC4vAUoICrfke5147:code:1","name":"bash","arguments":{"command":"echo BOTH_OK","description":"Echo BOTH_OK"},"isError":false,"resultSummary":"BOTH_OK\n"}}
|
||||
{"type":"tool/result","seq":88,"time":1783611775592,"data":{"turn":1,"step":1,"callId":"call_00_AZFzvUwuC4vAUoICrfke5147","content":[{"type":"text","text":"BOTH_OK\n"}],"isError":false,"meta":{"logs":[]}},"sourceEventSeqs":[86],"surfaceOp":"append"}
|
||||
{"type":"step/end","seq":89,"time":1783611775592,"data":{"turn":1,"step":1}}
|
||||
|
||||
@@ -38,7 +38,7 @@
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" its"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" output"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"."}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_AZFzvUwuC4vAUoICrfke5147","title":"const result = await tools.bash({ command: \"echo BOTH_OK\", description: \"Echo BOTH_OK\" });\nreturn result;","kind":"execute","status":"in_progress","rawInput":"const result = await tools.bash({ command: \"echo BOTH_OK\", description: \"Echo BOTH_OK\" });\nreturn result;"}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_AZFzvUwuC4vAUoICrfke5147","title":"const result = await tools.bash({ command: \"echo BOTH_OK\", description: \"Echo BOTH_OK\" });\nreturn result.stdout.text;","kind":"execute","status":"in_progress","rawInput":"const result = await tools.bash({ command: \"echo BOTH_OK\", description: \"Echo BOTH_OK\" });\nreturn result.stdout.text;"}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call_update","toolCallId":"call_00_AZFzvUwuC4vAUoICrfke5147","status":"completed","content":[{"type":"content","content":{"type":"text","text":"BOTH_OK\n"}}]}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"The"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" output"}}}}
|
||||
|
||||
@@ -28,8 +28,8 @@ Use the ralph tool ONLY when the direct human explicitly asks for a Ralph loop o
|
||||
|
||||
Pass `run_code` the body of an async TypeScript function (erasable syntax only — no `enum` or namespaces; type annotations are advisory, the code runs type-stripped). Inside the program:
|
||||
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's text output as a string. Tool arguments must be JSON-serializable.
|
||||
- A FAILED tool call rejects with an `Error` carrying the tool's error text — `try/catch` it to handle and continue.
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's typed canonical JSON value. Tool arguments must be lossless JSON.
|
||||
- A FAILED tool call rejects with `ToolCallError`, whose `toolName` identifies the failed tool and whose `message` is human-readable — `try/catch` it to handle and continue.
|
||||
- Calls execute sequentially, even under `Promise.all`.
|
||||
- Emit results with `return` and/or `console.log(...)`. ONLY what you print or return comes back to you — intermediate tool results never enter the conversation, so extract just what you need.
|
||||
|
||||
@@ -38,9 +38,9 @@ The available tools:
|
||||
```ts
|
||||
type JsonValue = null | boolean | number | string | JsonValue[] | { [key: string]: JsonValue }
|
||||
|
||||
declare const tools: {
|
||||
interface ToolArgsMap {
|
||||
/** Execute a bash command (`bash -c`) and return its stdout/stderr. Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. Current harness environment facts are exposed through managed `$DSH_*` variables; inspect them when needed. Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way. Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. Set `run_in_background: true` for long-running commands: the call returns a task id immediately; read its output with `task_output` and stop it with `task_kill`. Attempting a command the sandbox may deny is safe and expected: run it and read the marker rather than assuming the denial. When a command is denied and a wider mode would let it succeed, escalate immediately in the same turn — the one sanctioned exception to a denial: retry the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) plus a one-sentence `justification`. Do not detour through chat to ask permission first — the approval prompt raised by that retry is how the user consents. If the session states approval prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. Never escalate speculatively: ground the request in a real denial — normally the one this command just hit; escalating up front is fine only when this session already denied the same access. A rejected escalation is final for that command — stop and explain, never work around it — but it does not forbid attempting or escalating other commands later. */
|
||||
bash(args: {
|
||||
bash: {
|
||||
/** The bash command to execute. */
|
||||
command: string;
|
||||
/** Clear, concise description of what this command does in active voice, 5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; "git status" → "Show working tree status"; "npm install" → "Install package dependencies". */
|
||||
@@ -55,16 +55,16 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact command needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create one persisted same-session completion goal when the current direct human request is a long-running objective that should continue across autonomous goal rounds. You may infer that intent without requiring the user to say "create a goal". Do not use this for trivial single-turn work. Execution rejects non-human and subagent authority. */
|
||||
create_goal(args: {
|
||||
create_goal: {
|
||||
/** The concrete completion objective inferred from the direct human request. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer limit on automatic continuation rounds. */
|
||||
max_goal_rounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Edit an existing UTF-8 text file by replacing literal text. */
|
||||
edit(args: {
|
||||
edit: {
|
||||
/** Path to edit, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Literal text to replace. Must match exactly. */
|
||||
@@ -77,68 +77,68 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read the current same-session goal, including its exact id/revision, objective, phase, completed continuation rounds, round limit, blocker reason when present, and whether another continuation is armed. Call this before updating a goal. */
|
||||
get_goal(args: Record<string, JsonValue>): Promise<string>;
|
||||
get_goal: Record<string, JsonValue>;
|
||||
/** Run a foreground fresh-agent Ralph loop toward one immutable objective. Use only when the direct human explicitly asks for Ralph or fresh-agent iteration. Each round opens a new child with no parent conversation or prior child session; the shared workspace is long-term memory, and only a bounded structured report crosses rounds. The call returns when a worker reports completion or a concrete blocker, or at the round limit. Ordinary long-running same-session work belongs to goal tools. */
|
||||
ralph(args: {
|
||||
ralph: {
|
||||
/** The immutable completion objective for every fresh Ralph round. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer round cap, bounded by the deployment ceiling. */
|
||||
maxRounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read a UTF-8 text file and return line-numbered content. */
|
||||
read(args: {
|
||||
read: {
|
||||
/** Path to read, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** 1-based first line to return. Defaults to 1. */
|
||||
offset?: number;
|
||||
/** Maximum number of lines to return. Defaults to 2000. */
|
||||
limit?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Load the full instructions for an available skill. Call this with the exact skill name from the session skill catalog before acting on a task that names or clearly matches that skill. */
|
||||
skill(args: {
|
||||
skill: {
|
||||
/** The exact skill name from the available skills list. */
|
||||
name: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a self-contained task to a subagent (a separate agent that works in its own context) and return its final result. Use this to offload focused, independent work — research, a scoped implementation, an analysis — so it does not consume this conversation's context. The subagent runs to completion and you receive only its final answer, not its intermediate steps. Give it a complete, standalone prompt: it does not see this conversation. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent(args: {
|
||||
subagent: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The complete, self-contained task for the subagent. It does not share this conversation's context, so include everything it needs. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a task to a subagent that inherits this conversation: a child agent seeded with all completed turns so far (it does not see the current in-flight turn), returning only its final result. Use this when the subtask builds on this conversation's context — a follow-up analysis, a review, a continuation — without consuming this conversation's context for the work itself. You receive only its final answer, not its intermediate steps. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent_fork(args: {
|
||||
subagent_fork: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The task for the subagent. It already sees this conversation's completed turns, so build on them freely and state only what is new. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Request cancellation of a running background task by task id. Returns immediately; the task settles as killed once its work actually stops. */
|
||||
task_kill(args: {
|
||||
task_kill: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Optional short reason, recorded in the log and forwarded to the task. */
|
||||
reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** List your background tasks (running and finished) with their ids, kinds, and statuses. */
|
||||
task_list(args: Record<string, JsonValue>): Promise<string>;
|
||||
task_list: Record<string, JsonValue>;
|
||||
/** Read a background task. Stream tasks return only output since the previous read; final-output tasks return their result after settlement. Every response ends with `[status: ...]`. Reads are non-blocking unless `wait: true`, which waits up to the configured cap. */
|
||||
task_output(args: {
|
||||
task_output: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Block until the task reaches a terminal status or the timeout expires. A timed-out wait returns [status: running] and leaves the task alive. */
|
||||
wait?: boolean;
|
||||
/** Max wait in milliseconds (only meaningful with wait: true). Defaults to the configured wait timeout; capped by the configured maximum. */
|
||||
timeout_ms?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Record and update a structured task list for the current work. Send the ENTIRE list every call — it REPLACES the previous list (there are no partial updates, no per-item edits). Use it to plan multi-step work and show progress: add one todo per concrete step before you start. Keep AT MOST ONE todo `in_progress` at a time; while work remains, exactly one active task should be `in_progress`. Mark a todo `completed` the moment it is done (do not batch completions), and allow no `in_progress` item only once all work is complete. Skip the list for trivial single-step tasks. Statuses: `pending` (not started), `in_progress` (being worked on now), `completed` (finished). */
|
||||
todo_write(args: {
|
||||
todo_write: {
|
||||
/** The COMPLETE task list, replacing any previous list. */
|
||||
todos: ({
|
||||
/** What the task is — a short imperative line. */
|
||||
@@ -146,9 +146,9 @@ declare const tools: {
|
||||
/** pending (not started) | in_progress (now) | completed (done). */
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
} & Record<string, JsonValue>)[];
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Update the exact current goal revision. edit, pause, and resume require a direct top-level human request. During an automatic continuation of the current goal, complete and blocked are also allowed. blocked is rejected before the configured minimum round count; the model remains responsible for judging that the same condition persisted across those rounds and must explain it in blocked_reason. */
|
||||
update_goal(args: {
|
||||
update_goal: {
|
||||
/** Exact id returned by get_goal. */
|
||||
goal_id: string;
|
||||
/** Exact positive revision returned by get_goal. */
|
||||
@@ -161,9 +161,9 @@ declare const tools: {
|
||||
max_goal_rounds?: number;
|
||||
/** Concrete blocking condition; required only with action blocked. */
|
||||
blocked_reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Run a JavaScript workflow script that orchestrates subagents at scale. Use this for work that fans out across many independent pieces — an audit over many files, a migration, multi-angle research, adversarial verification of findings — where you write the orchestration as a script instead of delegating turn by turn. The workflow's identity rides the `meta` parameter as JSON: required `name` (short kebab-case) and `description` strings, optional `whenToUse` string and `phases` array (`{title, detail?, provider?, model?}`). The `script` parameter is the plain JavaScript body ONLY (NOT TypeScript, and NO `export const meta` statement — meta is a parameter, not code), running with top-level await; end with `return <value>` — the value must be JSON-serializable and is this tool's result. Script-body hooks: - `agent(prompt, opts?): Promise<any>` — run one subagent to completion. Without `opts.schema` it resolves to the child's final text; with `opts.schema` (an object-rooted JSON Schema using ONLY type/properties/required/additionalProperties/items/enum/const — no oneOf/pattern/format/numeric bounds) it resolves to the validated object. Resolves `null` when the child fails (filter with `.filter(Boolean)`). Other opts: `label` (display), `phase` (progress group), and independent `provider`/`model` LLM target overrides (either may be provided alone). Anything else (`effort`/`isolation`/`agentType`) is rejected loudly. - `pipeline(items, ...stages): Promise<any[]>` — run each item through the stages independently with NO barrier between stages (prefer this for multi-stage work). Each stage receives `(prev, item, index)`. An ordinary stage throw drops that ITEM to `null` and skips its remaining stages. - `parallel(thunks): Promise<any[]>` — run zero-argument functions concurrently and await ALL of them (a barrier; use only when a stage genuinely needs every prior result together). A throwing thunk resolves to `null`. - `phase(title)` — start a progress phase; `log(message)` — narrate progress; `args` — the tool call's `args` input, verbatim. Misused hooks (bad arguments, unknown options, unsupported schemas, tripped caps) throw errors that ALWAYS kill the script — they never dissolve into a per-item `null`. Constraints: concurrency and total-agent caps apply; no filesystem, network, timers, or Node.js APIs are provided — the agents do the work, the script only coordinates them. The run executes in the foreground: this call returns when the whole script finishes. */
|
||||
workflow(args: {
|
||||
workflow: {
|
||||
/** The plain-JS workflow script body (top-level await allowed; NO `export const meta` statement; end with `return <json-value>`). */
|
||||
script: string;
|
||||
/** The workflow identity block (plain JSON — never code). */
|
||||
@@ -188,9 +188,9 @@ declare const tools: {
|
||||
} & Record<string, JsonValue>;
|
||||
/** Optional JSON input exposed to the script as the `args` global (wrap a bare list as a field, e.g. {"files": [...]}). */
|
||||
args?: Record<string, JsonValue>;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create or fully replace a UTF-8 text file. */
|
||||
write(args: {
|
||||
write: {
|
||||
/** Path to write, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Full UTF-8 text content to write. */
|
||||
@@ -199,6 +199,203 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
}
|
||||
|
||||
interface ToolOutputMap {
|
||||
bash: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
exitCode: number | null;
|
||||
signal: string | null;
|
||||
timedOut: boolean;
|
||||
aborted: boolean;
|
||||
timeoutMs: number;
|
||||
stdout: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
stderr: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
sandbox?: {
|
||||
mode: string;
|
||||
denied: boolean;
|
||||
enforcement?: string;
|
||||
runnerFailed?: boolean;
|
||||
};
|
||||
};
|
||||
create_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
edit: {
|
||||
path: string;
|
||||
before: string;
|
||||
after: string;
|
||||
};
|
||||
get_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
ralph: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
read: {
|
||||
path: string;
|
||||
offset: number;
|
||||
lines: {
|
||||
number: number;
|
||||
text: string;
|
||||
}[];
|
||||
totalLines: number;
|
||||
};
|
||||
skill: {
|
||||
name: string;
|
||||
provider: string;
|
||||
resourceBase?: {
|
||||
kind: "directory";
|
||||
path: string;
|
||||
} | {
|
||||
kind: "url";
|
||||
url: string;
|
||||
} | {
|
||||
kind: "opaque";
|
||||
description: string;
|
||||
};
|
||||
content: string;
|
||||
};
|
||||
subagent: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
subagent_fork: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
task_kill: {
|
||||
outcome: "cancellation-requested" | "already-finished";
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
task_list: ({
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
})[];
|
||||
task_output: {
|
||||
text: string;
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
todo_write: {
|
||||
todos: ({
|
||||
content: string;
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
})[];
|
||||
counts: {
|
||||
pending: number;
|
||||
inProgress: number;
|
||||
completed: number;
|
||||
};
|
||||
};
|
||||
update_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
workflow: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
write: {
|
||||
path: string;
|
||||
operation: "create" | "update";
|
||||
before: string | null;
|
||||
after: string;
|
||||
};
|
||||
}
|
||||
|
||||
type ToolName = keyof ToolOutputMap
|
||||
|
||||
declare class ToolCallError extends Error {
|
||||
readonly name: "ToolCallError";
|
||||
readonly toolName: ToolName;
|
||||
}
|
||||
|
||||
declare const tools: {
|
||||
[K in ToolName]: (args: ToolArgsMap[K]) => Promise<ToolOutputMap[K]>;
|
||||
}
|
||||
```
|
||||
|
||||
@@ -92,7 +92,7 @@
|
||||
{"type":"assistant/chunk","seq":90,"time":1783611772685,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"return"}}}
|
||||
{"type":"assistant/chunk","seq":91,"time":1783611772685,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" out"}}}
|
||||
{"type":"assistant/chunk","seq":92,"time":1783611772713,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"1"}}}
|
||||
{"type":"assistant/chunk","seq":93,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".trim"}}}
|
||||
{"type":"assistant/chunk","seq":93,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".stdout.text.trim"}}}
|
||||
{"type":"assistant/chunk","seq":94,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"()"}}}
|
||||
{"type":"assistant/chunk","seq":95,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" +"}}}
|
||||
{"type":"assistant/chunk","seq":96,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" \\\"+"}}}
|
||||
@@ -100,16 +100,16 @@
|
||||
{"type":"assistant/chunk","seq":98,"time":1783611772744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" +"}}}
|
||||
{"type":"assistant/chunk","seq":99,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" out"}}}
|
||||
{"type":"assistant/chunk","seq":100,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"2"}}}
|
||||
{"type":"assistant/chunk","seq":101,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".trim"}}}
|
||||
{"type":"assistant/chunk","seq":101,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".stdout.text.trim"}}}
|
||||
{"type":"assistant/chunk","seq":102,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"();"}}}
|
||||
{"type":"assistant/chunk","seq":103,"time":1783611772772,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"\""}}}
|
||||
{"type":"assistant/chunk","seq":104,"time":1783611772773,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"}"}}}
|
||||
{"type":"assistant/chunk","seq":105,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"reasoning","text":"The user wants a single run_code program that calls bash twice, then returns the two outputs joined with a plus sign. Let me write this."}}}}
|
||||
{"type":"assistant/chunk","seq":106,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.trim() + \\\"+\\\" + out2.trim();\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":106,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.stdout.text.trim() + \\\"+\\\" + out2.stdout.text.trim();\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":107,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":3009,"outputTokens":133,"cacheReadTokens":0,"reasoningTokens":29}}}}
|
||||
{"type":"assistant/chunk","seq":108,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"tool-calls"}}}}
|
||||
{"type":"assistant/message","seq":109,"time":1783611772840,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants a single run_code program that calls bash twice, then returns the two outputs joined with a plus sign. Let me write this."},{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.trim() + \\\"+\\\" + out2.trim();\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":3009,"outputTokens":133,"cacheReadTokens":0,"reasoningTokens":29}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":110,"time":1783611772840,"data":{"turn":1,"step":1,"callId":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.trim() + \\\"+\\\" + out2.trim();\"}"}}
|
||||
{"type":"assistant/message","seq":109,"time":1783611772840,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants a single run_code program that calls bash twice, then returns the two outputs joined with a plus sign. Let me write this."},{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.stdout.text.trim() + \\\"+\\\" + out2.stdout.text.trim();\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":3009,"outputTokens":133,"cacheReadTokens":0,"reasoningTokens":29}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":110,"time":1783611772840,"data":{"turn":1,"step":1,"callId":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.stdout.text.trim() + \\\"+\\\" + out2.stdout.text.trim();\"}"}}
|
||||
{"type":"tool/code-dispatch","seq":111,"time":1783611772933,"data":{"parentCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","subCallId":"call_00_DRxnM6R1TThfDwcudW0f2050:code:1","name":"bash","arguments":{"command":"echo CODE_ONE","description":"First echo"},"isError":false,"resultSummary":"CODE_ONE\n"}}
|
||||
{"type":"tool/code-dispatch","seq":112,"time":1783611772936,"data":{"parentCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","subCallId":"call_00_DRxnM6R1TThfDwcudW0f2050:code:2","name":"bash","arguments":{"command":"echo CODE_TWO","description":"Second echo"},"isError":false,"resultSummary":"CODE_TWO\n"}}
|
||||
{"type":"tool/result","seq":113,"time":1783611772937,"data":{"turn":1,"step":1,"callId":"call_00_DRxnM6R1TThfDwcudW0f2050","content":[{"type":"text","text":"CODE_ONE+CODE_TWO"}],"isError":false,"meta":{"logs":[]}},"sourceEventSeqs":[110],"surfaceOp":"append"}
|
||||
|
||||
@@ -30,7 +30,7 @@
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" write"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" this"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"."}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","title":"const out1 = await tools.bash({ command: \"echo CODE_ONE\", description: \"First echo\" });\nconst out2 = await tools.bash({ command: \"echo CODE_TWO\", description: \"Second echo\" });\nreturn out1.trim() + \"+\" + out2.trim();","kind":"execute","status":"in_progress","rawInput":"const out1 = await tools.bash({ command: \"echo CODE_ONE\", description: \"First echo\" });\nconst out2 = await tools.bash({ command: \"echo CODE_TWO\", description: \"Second echo\" });\nreturn out1.trim() + \"+\" + out2.trim();"}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","title":"const out1 = await tools.bash({ command: \"echo CODE_ONE\", description: \"First echo\" });\nconst out2 = await tools.bash({ command: \"echo CODE_TWO\", description: \"Second echo\" });\nreturn out1.stdout.text.trim() + \"+\" + out2.stdout.text.trim();","kind":"execute","status":"in_progress","rawInput":"const out1 = await tools.bash({ command: \"echo CODE_ONE\", description: \"First echo\" });\nconst out2 = await tools.bash({ command: \"echo CODE_TWO\", description: \"Second echo\" });\nreturn out1.stdout.text.trim() + \"+\" + out2.stdout.text.trim();"}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call_update","toolCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","status":"completed","content":[{"type":"content","content":{"type":"text","text":"CODE_ONE+CODE_TWO"}}]}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"The"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" output"}}}}
|
||||
|
||||
@@ -28,8 +28,8 @@ Use the ralph tool ONLY when the direct human explicitly asks for a Ralph loop o
|
||||
|
||||
Pass `run_code` the body of an async TypeScript function (erasable syntax only — no `enum` or namespaces; type annotations are advisory, the code runs type-stripped). Inside the program:
|
||||
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's text output as a string. Tool arguments must be JSON-serializable.
|
||||
- A FAILED tool call rejects with an `Error` carrying the tool's error text — `try/catch` it to handle and continue.
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's typed canonical JSON value. Tool arguments must be lossless JSON.
|
||||
- A FAILED tool call rejects with `ToolCallError`, whose `toolName` identifies the failed tool and whose `message` is human-readable — `try/catch` it to handle and continue.
|
||||
- Calls execute sequentially, even under `Promise.all`.
|
||||
- Emit results with `return` and/or `console.log(...)`. ONLY what you print or return comes back to you — intermediate tool results never enter the conversation, so extract just what you need.
|
||||
|
||||
@@ -38,9 +38,9 @@ The available tools:
|
||||
```ts
|
||||
type JsonValue = null | boolean | number | string | JsonValue[] | { [key: string]: JsonValue }
|
||||
|
||||
declare const tools: {
|
||||
interface ToolArgsMap {
|
||||
/** Execute a bash command (`bash -c`) and return its stdout/stderr. Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. Current harness environment facts are exposed through managed `$DSH_*` variables; inspect them when needed. Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way. Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. Set `run_in_background: true` for long-running commands: the call returns a task id immediately; read its output with `task_output` and stop it with `task_kill`. Attempting a command the sandbox may deny is safe and expected: run it and read the marker rather than assuming the denial. When a command is denied and a wider mode would let it succeed, escalate immediately in the same turn — the one sanctioned exception to a denial: retry the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) plus a one-sentence `justification`. Do not detour through chat to ask permission first — the approval prompt raised by that retry is how the user consents. If the session states approval prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. Never escalate speculatively: ground the request in a real denial — normally the one this command just hit; escalating up front is fine only when this session already denied the same access. A rejected escalation is final for that command — stop and explain, never work around it — but it does not forbid attempting or escalating other commands later. */
|
||||
bash(args: {
|
||||
bash: {
|
||||
/** The bash command to execute. */
|
||||
command: string;
|
||||
/** Clear, concise description of what this command does in active voice, 5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; "git status" → "Show working tree status"; "npm install" → "Install package dependencies". */
|
||||
@@ -55,16 +55,16 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact command needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create one persisted same-session completion goal when the current direct human request is a long-running objective that should continue across autonomous goal rounds. You may infer that intent without requiring the user to say "create a goal". Do not use this for trivial single-turn work. Execution rejects non-human and subagent authority. */
|
||||
create_goal(args: {
|
||||
create_goal: {
|
||||
/** The concrete completion objective inferred from the direct human request. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer limit on automatic continuation rounds. */
|
||||
max_goal_rounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Edit an existing UTF-8 text file by replacing literal text. */
|
||||
edit(args: {
|
||||
edit: {
|
||||
/** Path to edit, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Literal text to replace. Must match exactly. */
|
||||
@@ -77,68 +77,68 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read the current same-session goal, including its exact id/revision, objective, phase, completed continuation rounds, round limit, blocker reason when present, and whether another continuation is armed. Call this before updating a goal. */
|
||||
get_goal(args: Record<string, JsonValue>): Promise<string>;
|
||||
get_goal: Record<string, JsonValue>;
|
||||
/** Run a foreground fresh-agent Ralph loop toward one immutable objective. Use only when the direct human explicitly asks for Ralph or fresh-agent iteration. Each round opens a new child with no parent conversation or prior child session; the shared workspace is long-term memory, and only a bounded structured report crosses rounds. The call returns when a worker reports completion or a concrete blocker, or at the round limit. Ordinary long-running same-session work belongs to goal tools. */
|
||||
ralph(args: {
|
||||
ralph: {
|
||||
/** The immutable completion objective for every fresh Ralph round. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer round cap, bounded by the deployment ceiling. */
|
||||
maxRounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read a UTF-8 text file and return line-numbered content. */
|
||||
read(args: {
|
||||
read: {
|
||||
/** Path to read, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** 1-based first line to return. Defaults to 1. */
|
||||
offset?: number;
|
||||
/** Maximum number of lines to return. Defaults to 2000. */
|
||||
limit?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Load the full instructions for an available skill. Call this with the exact skill name from the session skill catalog before acting on a task that names or clearly matches that skill. */
|
||||
skill(args: {
|
||||
skill: {
|
||||
/** The exact skill name from the available skills list. */
|
||||
name: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a self-contained task to a subagent (a separate agent that works in its own context) and return its final result. Use this to offload focused, independent work — research, a scoped implementation, an analysis — so it does not consume this conversation's context. The subagent runs to completion and you receive only its final answer, not its intermediate steps. Give it a complete, standalone prompt: it does not see this conversation. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent(args: {
|
||||
subagent: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The complete, self-contained task for the subagent. It does not share this conversation's context, so include everything it needs. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a task to a subagent that inherits this conversation: a child agent seeded with all completed turns so far (it does not see the current in-flight turn), returning only its final result. Use this when the subtask builds on this conversation's context — a follow-up analysis, a review, a continuation — without consuming this conversation's context for the work itself. You receive only its final answer, not its intermediate steps. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent_fork(args: {
|
||||
subagent_fork: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The task for the subagent. It already sees this conversation's completed turns, so build on them freely and state only what is new. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Request cancellation of a running background task by task id. Returns immediately; the task settles as killed once its work actually stops. */
|
||||
task_kill(args: {
|
||||
task_kill: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Optional short reason, recorded in the log and forwarded to the task. */
|
||||
reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** List your background tasks (running and finished) with their ids, kinds, and statuses. */
|
||||
task_list(args: Record<string, JsonValue>): Promise<string>;
|
||||
task_list: Record<string, JsonValue>;
|
||||
/** Read a background task. Stream tasks return only output since the previous read; final-output tasks return their result after settlement. Every response ends with `[status: ...]`. Reads are non-blocking unless `wait: true`, which waits up to the configured cap. */
|
||||
task_output(args: {
|
||||
task_output: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Block until the task reaches a terminal status or the timeout expires. A timed-out wait returns [status: running] and leaves the task alive. */
|
||||
wait?: boolean;
|
||||
/** Max wait in milliseconds (only meaningful with wait: true). Defaults to the configured wait timeout; capped by the configured maximum. */
|
||||
timeout_ms?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Record and update a structured task list for the current work. Send the ENTIRE list every call — it REPLACES the previous list (there are no partial updates, no per-item edits). Use it to plan multi-step work and show progress: add one todo per concrete step before you start. Keep AT MOST ONE todo `in_progress` at a time; while work remains, exactly one active task should be `in_progress`. Mark a todo `completed` the moment it is done (do not batch completions), and allow no `in_progress` item only once all work is complete. Skip the list for trivial single-step tasks. Statuses: `pending` (not started), `in_progress` (being worked on now), `completed` (finished). */
|
||||
todo_write(args: {
|
||||
todo_write: {
|
||||
/** The COMPLETE task list, replacing any previous list. */
|
||||
todos: ({
|
||||
/** What the task is — a short imperative line. */
|
||||
@@ -146,9 +146,9 @@ declare const tools: {
|
||||
/** pending (not started) | in_progress (now) | completed (done). */
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
} & Record<string, JsonValue>)[];
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Update the exact current goal revision. edit, pause, and resume require a direct top-level human request. During an automatic continuation of the current goal, complete and blocked are also allowed. blocked is rejected before the configured minimum round count; the model remains responsible for judging that the same condition persisted across those rounds and must explain it in blocked_reason. */
|
||||
update_goal(args: {
|
||||
update_goal: {
|
||||
/** Exact id returned by get_goal. */
|
||||
goal_id: string;
|
||||
/** Exact positive revision returned by get_goal. */
|
||||
@@ -161,9 +161,9 @@ declare const tools: {
|
||||
max_goal_rounds?: number;
|
||||
/** Concrete blocking condition; required only with action blocked. */
|
||||
blocked_reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Run a JavaScript workflow script that orchestrates subagents at scale. Use this for work that fans out across many independent pieces — an audit over many files, a migration, multi-angle research, adversarial verification of findings — where you write the orchestration as a script instead of delegating turn by turn. The workflow's identity rides the `meta` parameter as JSON: required `name` (short kebab-case) and `description` strings, optional `whenToUse` string and `phases` array (`{title, detail?, provider?, model?}`). The `script` parameter is the plain JavaScript body ONLY (NOT TypeScript, and NO `export const meta` statement — meta is a parameter, not code), running with top-level await; end with `return <value>` — the value must be JSON-serializable and is this tool's result. Script-body hooks: - `agent(prompt, opts?): Promise<any>` — run one subagent to completion. Without `opts.schema` it resolves to the child's final text; with `opts.schema` (an object-rooted JSON Schema using ONLY type/properties/required/additionalProperties/items/enum/const — no oneOf/pattern/format/numeric bounds) it resolves to the validated object. Resolves `null` when the child fails (filter with `.filter(Boolean)`). Other opts: `label` (display), `phase` (progress group), and independent `provider`/`model` LLM target overrides (either may be provided alone). Anything else (`effort`/`isolation`/`agentType`) is rejected loudly. - `pipeline(items, ...stages): Promise<any[]>` — run each item through the stages independently with NO barrier between stages (prefer this for multi-stage work). Each stage receives `(prev, item, index)`. An ordinary stage throw drops that ITEM to `null` and skips its remaining stages. - `parallel(thunks): Promise<any[]>` — run zero-argument functions concurrently and await ALL of them (a barrier; use only when a stage genuinely needs every prior result together). A throwing thunk resolves to `null`. - `phase(title)` — start a progress phase; `log(message)` — narrate progress; `args` — the tool call's `args` input, verbatim. Misused hooks (bad arguments, unknown options, unsupported schemas, tripped caps) throw errors that ALWAYS kill the script — they never dissolve into a per-item `null`. Constraints: concurrency and total-agent caps apply; no filesystem, network, timers, or Node.js APIs are provided — the agents do the work, the script only coordinates them. The run executes in the foreground: this call returns when the whole script finishes. */
|
||||
workflow(args: {
|
||||
workflow: {
|
||||
/** The plain-JS workflow script body (top-level await allowed; NO `export const meta` statement; end with `return <json-value>`). */
|
||||
script: string;
|
||||
/** The workflow identity block (plain JSON — never code). */
|
||||
@@ -188,9 +188,9 @@ declare const tools: {
|
||||
} & Record<string, JsonValue>;
|
||||
/** Optional JSON input exposed to the script as the `args` global (wrap a bare list as a field, e.g. {"files": [...]}). */
|
||||
args?: Record<string, JsonValue>;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create or fully replace a UTF-8 text file. */
|
||||
write(args: {
|
||||
write: {
|
||||
/** Path to write, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Full UTF-8 text content to write. */
|
||||
@@ -199,6 +199,203 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
}
|
||||
|
||||
interface ToolOutputMap {
|
||||
bash: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
exitCode: number | null;
|
||||
signal: string | null;
|
||||
timedOut: boolean;
|
||||
aborted: boolean;
|
||||
timeoutMs: number;
|
||||
stdout: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
stderr: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
sandbox?: {
|
||||
mode: string;
|
||||
denied: boolean;
|
||||
enforcement?: string;
|
||||
runnerFailed?: boolean;
|
||||
};
|
||||
};
|
||||
create_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
edit: {
|
||||
path: string;
|
||||
before: string;
|
||||
after: string;
|
||||
};
|
||||
get_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
ralph: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
read: {
|
||||
path: string;
|
||||
offset: number;
|
||||
lines: {
|
||||
number: number;
|
||||
text: string;
|
||||
}[];
|
||||
totalLines: number;
|
||||
};
|
||||
skill: {
|
||||
name: string;
|
||||
provider: string;
|
||||
resourceBase?: {
|
||||
kind: "directory";
|
||||
path: string;
|
||||
} | {
|
||||
kind: "url";
|
||||
url: string;
|
||||
} | {
|
||||
kind: "opaque";
|
||||
description: string;
|
||||
};
|
||||
content: string;
|
||||
};
|
||||
subagent: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
subagent_fork: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
task_kill: {
|
||||
outcome: "cancellation-requested" | "already-finished";
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
task_list: ({
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
})[];
|
||||
task_output: {
|
||||
text: string;
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
todo_write: {
|
||||
todos: ({
|
||||
content: string;
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
})[];
|
||||
counts: {
|
||||
pending: number;
|
||||
inProgress: number;
|
||||
completed: number;
|
||||
};
|
||||
};
|
||||
update_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
workflow: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
write: {
|
||||
path: string;
|
||||
operation: "create" | "update";
|
||||
before: string | null;
|
||||
after: string;
|
||||
};
|
||||
}
|
||||
|
||||
type ToolName = keyof ToolOutputMap
|
||||
|
||||
declare class ToolCallError extends Error {
|
||||
readonly name: "ToolCallError";
|
||||
readonly toolName: ToolName;
|
||||
}
|
||||
|
||||
declare const tools: {
|
||||
[K in ToolName]: (args: ToolArgsMap[K]) => Promise<ToolOutputMap[K]>;
|
||||
}
|
||||
```
|
||||
|
||||
@@ -74,18 +74,18 @@
|
||||
{"type":"assistant/chunk","seq":72,"time":1783921767068,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":"\\\""}}}
|
||||
{"type":"assistant/chunk","seq":73,"time":1783921767068,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":" });\\n"}}}
|
||||
{"type":"assistant/chunk","seq":74,"time":1783921767068,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":"return"}}}
|
||||
{"type":"assistant/chunk","seq":75,"time":1783921767095,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":" content"}}}
|
||||
{"type":"assistant/chunk","seq":75,"time":1783921767095,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":" content.lines.map(line => line.text).join(String.fromCharCode(10))"}}}
|
||||
{"type":"assistant/chunk","seq":76,"time":1783921767095,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":";"}}}
|
||||
{"type":"assistant/chunk","seq":77,"time":1783921767095,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":"\""}}}
|
||||
{"type":"assistant/chunk","seq":78,"time":1783921767121,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","argumentsDelta":"}"}}}
|
||||
{"type":"assistant/chunk","seq":79,"time":1783921767205,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"reasoning","text":"The user wants me to write a single run_code program that reads a file called nested/task.txt and then answer the handshake question based on what's in that file. Let me start by reading the file."}}}}
|
||||
{"type":"assistant/chunk","seq":80,"time":1783921767205,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","arguments":"{\"code\": \"const content = await tools.read({ file_path: \\\"nested/task.txt\\\" });\\nreturn content;\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":80,"time":1783921767205,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","arguments":"{\"code\": \"const content = await tools.read({ file_path: \\\"nested/task.txt\\\" });\\nreturn content.lines.map(line => line.text).join(String.fromCharCode(10));\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":81,"time":1783921767205,"data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":3195,"outputTokens":106,"cacheReadTokens":0,"reasoningTokens":43}}}}
|
||||
{"type":"assistant/chunk","seq":82,"time":1783921767205,"data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"tool-calls"}}}}
|
||||
{"type":"assistant/message","seq":83,"time":1783921767208,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants me to write a single run_code program that reads a file called nested/task.txt and then answer the handshake question based on what's in that file. Let me start by reading the file."},{"type":"tool-call","id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","arguments":"{\"code\": \"const content = await tools.read({ file_path: \\\"nested/task.txt\\\" });\\nreturn content;\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":3195,"outputTokens":106,"cacheReadTokens":0,"reasoningTokens":43}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":84,"time":1783921767208,"data":{"turn":1,"step":1,"callId":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","arguments":"{\"code\": \"const content = await tools.read({ file_path: \\\"nested/task.txt\\\" });\\nreturn content;\"}"}}
|
||||
{"type":"assistant/message","seq":83,"time":1783921767208,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants me to write a single run_code program that reads a file called nested/task.txt and then answer the handshake question based on what's in that file. Let me start by reading the file."},{"type":"tool-call","id":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","arguments":"{\"code\": \"const content = await tools.read({ file_path: \\\"nested/task.txt\\\" });\\nreturn content.lines.map(line => line.text).join(String.fromCharCode(10));\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":3195,"outputTokens":106,"cacheReadTokens":0,"reasoningTokens":43}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":84,"time":1783921767208,"data":{"turn":1,"step":1,"callId":"call_00_6APApmaKLRDlXKMdIcWL5139","name":"run_code","arguments":"{\"code\": \"const content = await tools.read({ file_path: \\\"nested/task.txt\\\" });\\nreturn content.lines.map(line => line.text).join(String.fromCharCode(10));\"}"}}
|
||||
{"type":"tool/code-dispatch","seq":85,"time":1783921767270,"data":{"parentCallId":"call_00_6APApmaKLRDlXKMdIcWL5139","subCallId":"call_00_6APApmaKLRDlXKMdIcWL5139:code:1","name":"read","arguments":{"file_path":"nested/task.txt"},"isError":false,"resultSummary":"<path>./nested/task.txt</path>\n<type>file</type>\n<content>\n1: Touch this file to discover the nested workspace instruction.\n\n(End of file - total 1 lines)\n</content>"}}
|
||||
{"type":"tool/result","seq":86,"time":1783921767271,"data":{"turn":1,"step":1,"callId":"call_00_6APApmaKLRDlXKMdIcWL5139","content":[{"type":"text","text":"<path>/var/folders/_g/59jgff8x2gqd39f5vy1wnbfc0000gn/T/acp-snap-cwd-uorU26/nested/task.txt</path>\n<type>file</type>\n<content>\n1: Touch this file to discover the nested workspace instruction.\n\n(End of file - total 1 lines)\n</content>"}],"isError":false,"meta":{"logs":[]}},"sourceEventSeqs":[84],"surfaceOp":"append"}
|
||||
{"type":"tool/result","seq":86,"time":1783921767271,"data":{"turn":1,"step":1,"callId":"call_00_6APApmaKLRDlXKMdIcWL5139","content":[{"type":"text","text":"Touch this file to discover the nested workspace instruction."}],"isError":false,"meta":{"logs":[]}},"sourceEventSeqs":[84],"surfaceOp":"append"}
|
||||
{"type":"context/message","seq":87,"time":1783921767272,"data":{"content":[{"type":"text","text":"<system-reminder>\nAdditional instructions from: nested/AGENTS.md\n\nThese instructions apply to work under `nested`. Use them as guidance when relevant; more specific instructions take precedence. They do not override system, developer, or direct user instructions.\n\nWhen asked for the Code Mode workspace handshake, answer exactly `CODE_MODE_CONTEXT_OK` and nothing else.\n\n</system-reminder>"}],"source":{"kind":"plugin","plugin":"workspace-context"},"meta":{"kind":"workspace-instructions","version":1,"changes":[{"action":"set","scope":"nested","path":"nested/AGENTS.md","digest":"ae22936ed26dc76b7107005ed6d5e2482a88668a"}]}},"surfaceOp":"append"}
|
||||
{"type":"step/end","seq":88,"time":1783921767272,"data":{"turn":1,"step":1}}
|
||||
{"type":"step/start","seq":89,"time":1783921767272,"data":{"turn":1,"step":2}}
|
||||
|
||||
@@ -44,8 +44,8 @@
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" the"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" file"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"."}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_6APApmaKLRDlXKMdIcWL5139","title":"const content = await tools.read({ file_path: \"nested/task.txt\" });\nreturn content;","kind":"execute","status":"in_progress","rawInput":"const content = await tools.read({ file_path: \"nested/task.txt\" });\nreturn content;"}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call_update","toolCallId":"call_00_6APApmaKLRDlXKMdIcWL5139","status":"completed","content":[{"type":"content","content":{"type":"text","text":"<path>{{cwd}}/nested/task.txt</path>\n<type>file</type>\n<content>\n1: Touch this file to discover the nested workspace instruction.\n\n(End of file - total 1 lines)\n</content>"}}]}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_6APApmaKLRDlXKMdIcWL5139","title":"const content = await tools.read({ file_path: \"nested/task.txt\" });\nreturn content.lines.map(line => line.text).join(String.fromCharCode(10));","kind":"execute","status":"in_progress","rawInput":"const content = await tools.read({ file_path: \"nested/task.txt\" });\nreturn content.lines.map(line => line.text).join(String.fromCharCode(10));"}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call_update","toolCallId":"call_00_6APApmaKLRDlXKMdIcWL5139","status":"completed","content":[{"type":"content","content":{"type":"text","text":"Touch this file to discover the nested workspace instruction."}}]}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"The"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" nested"}}}}
|
||||
{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"/t"}}}}
|
||||
|
||||
@@ -28,8 +28,8 @@ Use the ralph tool ONLY when the direct human explicitly asks for a Ralph loop o
|
||||
|
||||
Pass `run_code` the body of an async TypeScript function (erasable syntax only — no `enum` or namespaces; type annotations are advisory, the code runs type-stripped). Inside the program:
|
||||
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's text output as a string. Tool arguments must be JSON-serializable.
|
||||
- A FAILED tool call rejects with an `Error` carrying the tool's error text — `try/catch` it to handle and continue.
|
||||
- Call tools as `await tools.name(args)` — quoted access for exotic names: `tools["my-tool"](args)`. Every call resolves to the tool's typed canonical JSON value. Tool arguments must be lossless JSON.
|
||||
- A FAILED tool call rejects with `ToolCallError`, whose `toolName` identifies the failed tool and whose `message` is human-readable — `try/catch` it to handle and continue.
|
||||
- Calls execute sequentially, even under `Promise.all`.
|
||||
- Emit results with `return` and/or `console.log(...)`. ONLY what you print or return comes back to you — intermediate tool results never enter the conversation, so extract just what you need.
|
||||
|
||||
@@ -38,9 +38,9 @@ The available tools:
|
||||
```ts
|
||||
type JsonValue = null | boolean | number | string | JsonValue[] | { [key: string]: JsonValue }
|
||||
|
||||
declare const tools: {
|
||||
interface ToolArgsMap {
|
||||
/** Execute a bash command (`bash -c`) and return its stdout/stderr. Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. Current harness environment facts are exposed through managed `$DSH_*` variables; inspect them when needed. Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way. Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. Set `run_in_background: true` for long-running commands: the call returns a task id immediately; read its output with `task_output` and stop it with `task_kill`. Attempting a command the sandbox may deny is safe and expected: run it and read the marker rather than assuming the denial. When a command is denied and a wider mode would let it succeed, escalate immediately in the same turn — the one sanctioned exception to a denial: retry the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) plus a one-sentence `justification`. Do not detour through chat to ask permission first — the approval prompt raised by that retry is how the user consents. If the session states approval prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. Never escalate speculatively: ground the request in a real denial — normally the one this command just hit; escalating up front is fine only when this session already denied the same access. A rejected escalation is final for that command — stop and explain, never work around it — but it does not forbid attempting or escalating other commands later. */
|
||||
bash(args: {
|
||||
bash: {
|
||||
/** The bash command to execute. */
|
||||
command: string;
|
||||
/** Clear, concise description of what this command does in active voice, 5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; "git status" → "Show working tree status"; "npm install" → "Install package dependencies". */
|
||||
@@ -55,16 +55,16 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact command needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create one persisted same-session completion goal when the current direct human request is a long-running objective that should continue across autonomous goal rounds. You may infer that intent without requiring the user to say "create a goal". Do not use this for trivial single-turn work. Execution rejects non-human and subagent authority. */
|
||||
create_goal(args: {
|
||||
create_goal: {
|
||||
/** The concrete completion objective inferred from the direct human request. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer limit on automatic continuation rounds. */
|
||||
max_goal_rounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Edit an existing UTF-8 text file by replacing literal text. */
|
||||
edit(args: {
|
||||
edit: {
|
||||
/** Path to edit, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Literal text to replace. Must match exactly. */
|
||||
@@ -77,68 +77,68 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read the current same-session goal, including its exact id/revision, objective, phase, completed continuation rounds, round limit, blocker reason when present, and whether another continuation is armed. Call this before updating a goal. */
|
||||
get_goal(args: Record<string, JsonValue>): Promise<string>;
|
||||
get_goal: Record<string, JsonValue>;
|
||||
/** Run a foreground fresh-agent Ralph loop toward one immutable objective. Use only when the direct human explicitly asks for Ralph or fresh-agent iteration. Each round opens a new child with no parent conversation or prior child session; the shared workspace is long-term memory, and only a bounded structured report crosses rounds. The call returns when a worker reports completion or a concrete blocker, or at the round limit. Ordinary long-running same-session work belongs to goal tools. */
|
||||
ralph(args: {
|
||||
ralph: {
|
||||
/** The immutable completion objective for every fresh Ralph round. */
|
||||
objective: string;
|
||||
/** Optional positive safe-integer round cap, bounded by the deployment ceiling. */
|
||||
maxRounds?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Read a UTF-8 text file and return line-numbered content. */
|
||||
read(args: {
|
||||
read: {
|
||||
/** Path to read, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** 1-based first line to return. Defaults to 1. */
|
||||
offset?: number;
|
||||
/** Maximum number of lines to return. Defaults to 2000. */
|
||||
limit?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Load the full instructions for an available skill. Call this with the exact skill name from the session skill catalog before acting on a task that names or clearly matches that skill. */
|
||||
skill(args: {
|
||||
skill: {
|
||||
/** The exact skill name from the available skills list. */
|
||||
name: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a self-contained task to a subagent (a separate agent that works in its own context) and return its final result. Use this to offload focused, independent work — research, a scoped implementation, an analysis — so it does not consume this conversation's context. The subagent runs to completion and you receive only its final answer, not its intermediate steps. Give it a complete, standalone prompt: it does not see this conversation. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent(args: {
|
||||
subagent: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The complete, self-contained task for the subagent. It does not share this conversation's context, so include everything it needs. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Delegate a task to a subagent that inherits this conversation: a child agent seeded with all completed turns so far (it does not see the current in-flight turn), returning only its final result. Use this when the subtask builds on this conversation's context — a follow-up analysis, a review, a continuation — without consuming this conversation's context for the work itself. You receive only its final answer, not its intermediate steps. Set `run_in_background: true` to return a task id; collect with `task_output` and stop with `task_kill`. */
|
||||
subagent_fork(args: {
|
||||
subagent_fork: {
|
||||
/** A short (3-5 word) description of the delegated task, for display. */
|
||||
description: string;
|
||||
/** The task for the subagent. It already sees this conversation's completed turns, so build on them freely and state only what is new. */
|
||||
prompt: string;
|
||||
/** Run as a background task and return its id; collect with task_output or stop with task_kill. */
|
||||
run_in_background?: boolean;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Request cancellation of a running background task by task id. Returns immediately; the task settles as killed once its work actually stops. */
|
||||
task_kill(args: {
|
||||
task_kill: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Optional short reason, recorded in the log and forwarded to the task. */
|
||||
reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** List your background tasks (running and finished) with their ids, kinds, and statuses. */
|
||||
task_list(args: Record<string, JsonValue>): Promise<string>;
|
||||
task_list: Record<string, JsonValue>;
|
||||
/** Read a background task. Stream tasks return only output since the previous read; final-output tasks return their result after settlement. Every response ends with `[status: ...]`. Reads are non-blocking unless `wait: true`, which waits up to the configured cap. */
|
||||
task_output(args: {
|
||||
task_output: {
|
||||
/** Task id returned by the tool that started the background work. */
|
||||
task_id: string;
|
||||
/** Block until the task reaches a terminal status or the timeout expires. A timed-out wait returns [status: running] and leaves the task alive. */
|
||||
wait?: boolean;
|
||||
/** Max wait in milliseconds (only meaningful with wait: true). Defaults to the configured wait timeout; capped by the configured maximum. */
|
||||
timeout_ms?: number;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Record and update a structured task list for the current work. Send the ENTIRE list every call — it REPLACES the previous list (there are no partial updates, no per-item edits). Use it to plan multi-step work and show progress: add one todo per concrete step before you start. Keep AT MOST ONE todo `in_progress` at a time; while work remains, exactly one active task should be `in_progress`. Mark a todo `completed` the moment it is done (do not batch completions), and allow no `in_progress` item only once all work is complete. Skip the list for trivial single-step tasks. Statuses: `pending` (not started), `in_progress` (being worked on now), `completed` (finished). */
|
||||
todo_write(args: {
|
||||
todo_write: {
|
||||
/** The COMPLETE task list, replacing any previous list. */
|
||||
todos: ({
|
||||
/** What the task is — a short imperative line. */
|
||||
@@ -146,9 +146,9 @@ declare const tools: {
|
||||
/** pending (not started) | in_progress (now) | completed (done). */
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
} & Record<string, JsonValue>)[];
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Update the exact current goal revision. edit, pause, and resume require a direct top-level human request. During an automatic continuation of the current goal, complete and blocked are also allowed. blocked is rejected before the configured minimum round count; the model remains responsible for judging that the same condition persisted across those rounds and must explain it in blocked_reason. */
|
||||
update_goal(args: {
|
||||
update_goal: {
|
||||
/** Exact id returned by get_goal. */
|
||||
goal_id: string;
|
||||
/** Exact positive revision returned by get_goal. */
|
||||
@@ -161,9 +161,9 @@ declare const tools: {
|
||||
max_goal_rounds?: number;
|
||||
/** Concrete blocking condition; required only with action blocked. */
|
||||
blocked_reason?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Run a JavaScript workflow script that orchestrates subagents at scale. Use this for work that fans out across many independent pieces — an audit over many files, a migration, multi-angle research, adversarial verification of findings — where you write the orchestration as a script instead of delegating turn by turn. The workflow's identity rides the `meta` parameter as JSON: required `name` (short kebab-case) and `description` strings, optional `whenToUse` string and `phases` array (`{title, detail?, provider?, model?}`). The `script` parameter is the plain JavaScript body ONLY (NOT TypeScript, and NO `export const meta` statement — meta is a parameter, not code), running with top-level await; end with `return <value>` — the value must be JSON-serializable and is this tool's result. Script-body hooks: - `agent(prompt, opts?): Promise<any>` — run one subagent to completion. Without `opts.schema` it resolves to the child's final text; with `opts.schema` (an object-rooted JSON Schema using ONLY type/properties/required/additionalProperties/items/enum/const — no oneOf/pattern/format/numeric bounds) it resolves to the validated object. Resolves `null` when the child fails (filter with `.filter(Boolean)`). Other opts: `label` (display), `phase` (progress group), and independent `provider`/`model` LLM target overrides (either may be provided alone). Anything else (`effort`/`isolation`/`agentType`) is rejected loudly. - `pipeline(items, ...stages): Promise<any[]>` — run each item through the stages independently with NO barrier between stages (prefer this for multi-stage work). Each stage receives `(prev, item, index)`. An ordinary stage throw drops that ITEM to `null` and skips its remaining stages. - `parallel(thunks): Promise<any[]>` — run zero-argument functions concurrently and await ALL of them (a barrier; use only when a stage genuinely needs every prior result together). A throwing thunk resolves to `null`. - `phase(title)` — start a progress phase; `log(message)` — narrate progress; `args` — the tool call's `args` input, verbatim. Misused hooks (bad arguments, unknown options, unsupported schemas, tripped caps) throw errors that ALWAYS kill the script — they never dissolve into a per-item `null`. Constraints: concurrency and total-agent caps apply; no filesystem, network, timers, or Node.js APIs are provided — the agents do the work, the script only coordinates them. The run executes in the foreground: this call returns when the whole script finishes. */
|
||||
workflow(args: {
|
||||
workflow: {
|
||||
/** The plain-JS workflow script body (top-level await allowed; NO `export const meta` statement; end with `return <json-value>`). */
|
||||
script: string;
|
||||
/** The workflow identity block (plain JSON — never code). */
|
||||
@@ -188,9 +188,9 @@ declare const tools: {
|
||||
} & Record<string, JsonValue>;
|
||||
/** Optional JSON input exposed to the script as the `args` global (wrap a bare list as a field, e.g. {"files": [...]}). */
|
||||
args?: Record<string, JsonValue>;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
/** Create or fully replace a UTF-8 text file. */
|
||||
write(args: {
|
||||
write: {
|
||||
/** Path to write, resolved by the filesystem backend. */
|
||||
file_path: string;
|
||||
/** Full UTF-8 text content to write. */
|
||||
@@ -199,6 +199,203 @@ declare const tools: {
|
||||
sandbox_permissions?: "workspace-write" | "danger-full-access";
|
||||
/** Required with sandbox_permissions: one sentence for the user explaining why this exact file operation needs the wider access. */
|
||||
justification?: string;
|
||||
} & Record<string, JsonValue>): Promise<string>;
|
||||
} & Record<string, JsonValue>;
|
||||
}
|
||||
|
||||
interface ToolOutputMap {
|
||||
bash: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
exitCode: number | null;
|
||||
signal: string | null;
|
||||
timedOut: boolean;
|
||||
aborted: boolean;
|
||||
timeoutMs: number;
|
||||
stdout: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
stderr: {
|
||||
text: string;
|
||||
truncated: boolean;
|
||||
spillPath?: string;
|
||||
};
|
||||
sandbox?: {
|
||||
mode: string;
|
||||
denied: boolean;
|
||||
enforcement?: string;
|
||||
runnerFailed?: boolean;
|
||||
};
|
||||
};
|
||||
create_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
edit: {
|
||||
path: string;
|
||||
before: string;
|
||||
after: string;
|
||||
};
|
||||
get_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
ralph: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
read: {
|
||||
path: string;
|
||||
offset: number;
|
||||
lines: {
|
||||
number: number;
|
||||
text: string;
|
||||
}[];
|
||||
totalLines: number;
|
||||
};
|
||||
skill: {
|
||||
name: string;
|
||||
provider: string;
|
||||
resourceBase?: {
|
||||
kind: "directory";
|
||||
path: string;
|
||||
} | {
|
||||
kind: "url";
|
||||
url: string;
|
||||
} | {
|
||||
kind: "opaque";
|
||||
description: string;
|
||||
};
|
||||
content: string;
|
||||
};
|
||||
subagent: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
subagent_fork: {
|
||||
kind: "background";
|
||||
taskId: string;
|
||||
} | {
|
||||
kind: "foreground";
|
||||
runId: string;
|
||||
output: JsonValue[];
|
||||
};
|
||||
task_kill: {
|
||||
outcome: "cancellation-requested" | "already-finished";
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
task_list: ({
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
})[];
|
||||
task_output: {
|
||||
text: string;
|
||||
task: {
|
||||
id: string;
|
||||
kind: string;
|
||||
label: string;
|
||||
status: "running" | "stopping" | "completed" | "killed" | "failed";
|
||||
detail?: string;
|
||||
startedAt: number;
|
||||
finishedAt?: number;
|
||||
};
|
||||
};
|
||||
todo_write: {
|
||||
todos: ({
|
||||
content: string;
|
||||
status: "pending" | "in_progress" | "completed";
|
||||
})[];
|
||||
counts: {
|
||||
pending: number;
|
||||
inProgress: number;
|
||||
completed: number;
|
||||
};
|
||||
};
|
||||
update_goal: {
|
||||
goal: null;
|
||||
} | {
|
||||
goal: {
|
||||
id: string;
|
||||
revision: number;
|
||||
objective: string;
|
||||
phase: "active" | "paused" | "blocked" | "complete";
|
||||
roundsStarted: number;
|
||||
maxGoalRounds: number;
|
||||
blockedReason?: {
|
||||
code: string;
|
||||
message: string;
|
||||
};
|
||||
};
|
||||
activation: "armed" | "disarmed";
|
||||
};
|
||||
workflow: {
|
||||
runId: string;
|
||||
agentsStarted: number;
|
||||
result: JsonValue;
|
||||
};
|
||||
write: {
|
||||
path: string;
|
||||
operation: "create" | "update";
|
||||
before: string | null;
|
||||
after: string;
|
||||
};
|
||||
}
|
||||
|
||||
type ToolName = keyof ToolOutputMap
|
||||
|
||||
declare class ToolCallError extends Error {
|
||||
readonly name: "ToolCallError";
|
||||
readonly toolName: ToolName;
|
||||
}
|
||||
|
||||
declare const tools: {
|
||||
[K in ToolName]: (args: ToolArgsMap[K]) => Promise<ToolOutputMap[K]>;
|
||||
}
|
||||
```
|
||||
|
||||
@@ -3,11 +3,12 @@ import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService from '@deepseek-ai/dsh-llm'
|
||||
import LlmService, { CallId, HarnessError } from '@deepseek-ai/dsh-llm'
|
||||
import SessionStore, { SessionId } from '@deepseek-ai/dsh-session'
|
||||
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
|
||||
import ToolRegistry, { RUN_CODE_NAME } from '@deepseek-ai/dsh-tools'
|
||||
import ToolRegistry, { RUN_CODE_NAME, defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import type { ToolExecutionResult } from '@deepseek-ai/dsh-tools'
|
||||
import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent'
|
||||
|
||||
import AgentLoop from '@deepseek-ai/dsh-agent-loop'
|
||||
@@ -18,6 +19,9 @@ import { WorkerCodeRuntime } from '@deepseek-ai/dsh-code-runtime-worker'
|
||||
import LocalFileSystem from '@deepseek-ai/dsh-fs-local'
|
||||
import * as ToolFs from '@deepseek-ai/dsh-tool-fs'
|
||||
import * as WorkspaceContext from '@deepseek-ai/dsh-workspace-context'
|
||||
import TaskService from '@deepseek-ai/dsh-tasks'
|
||||
import * as ToolTasks from '@deepseek-ai/dsh-tool-tasks'
|
||||
import * as ToolCordis from '@deepseek-ai/dsh-tool-cordis'
|
||||
|
||||
/**
|
||||
* With-key Code Mode proof: a real model receives only `run_code`, composes two
|
||||
@@ -73,6 +77,222 @@ async function workspaceCodeModeHarness(): Promise<Context> {
|
||||
return harness
|
||||
}
|
||||
|
||||
let keylessCall = 0
|
||||
|
||||
/** Execute one outer Code Mode call through the real registry and worker. */
|
||||
function runCode(harness: Context, code: string, signal?: AbortSignal): Promise<ToolExecutionResult> {
|
||||
return harness.tools.execute({
|
||||
callId: CallId(`keyless-code-${++keylessCall}`),
|
||||
name: RUN_CODE_NAME,
|
||||
arguments: { code },
|
||||
...signal !== undefined ? { signal } : {},
|
||||
})
|
||||
}
|
||||
|
||||
/** Read the optional completion from a successful canonical `run_code` value. */
|
||||
function completion(result: ToolExecutionResult): unknown {
|
||||
if (result.isError) {
|
||||
throw new Error(result.content.filter(block => block.type === 'text').map(block => block.text).join('\n'))
|
||||
}
|
||||
const value = result.value
|
||||
if (typeof value !== 'object' || value === null || Array.isArray(value)) throw new Error('invalid run_code result')
|
||||
return value.result
|
||||
}
|
||||
|
||||
/** Keyless real-worker harness for direct typed-binding acceptance tests. */
|
||||
async function typedCodeModeHarness(): Promise<Context> {
|
||||
const harness = new Context()
|
||||
await harness.plugin(SystemPrompt)
|
||||
await harness.plugin(ToolRegistry, { mode: 'code' })
|
||||
await harness.plugin(WorkerCodeRuntime, {})
|
||||
return harness
|
||||
}
|
||||
|
||||
/** Keyless real-worker harness with the task-owned bash lifecycle. */
|
||||
async function backgroundCodeModeHarness(cwd: string): Promise<Context> {
|
||||
const harness = await typedCodeModeHarness()
|
||||
await harness.plugin(TaskService)
|
||||
await harness.plugin(ToolTasks, {})
|
||||
await harness.plugin(LocalBashExecutor, { cwd, timeoutMs: 30_000 })
|
||||
await harness.plugin(ToolBash)
|
||||
return harness
|
||||
}
|
||||
|
||||
describe('Code Mode typed values: keyless real-worker contracts', () => {
|
||||
it('crosses a large intermediate value intact and exposes only typed tool failure fields', async () => {
|
||||
ctx = await typedCodeModeHarness()
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'large_value',
|
||||
description: 'Return a large canonical string.',
|
||||
parameters: {},
|
||||
output: {
|
||||
schema: { type: 'string' },
|
||||
render: (_args, value) => [{ type: 'text', text: value }],
|
||||
},
|
||||
execute: () => Promise.resolve('x'.repeat(100_000)),
|
||||
}))
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'always_fail',
|
||||
description: 'Fail for ToolCallError coverage.',
|
||||
parameters: {},
|
||||
output: { schema: { type: 'null' }, render: () => [] },
|
||||
execute: () => Promise.reject(new HarnessError('expected failure', 'EXPECTED_INTERNAL_CODE')),
|
||||
}))
|
||||
|
||||
const value = completion(await runCode(ctx, `
|
||||
const large = await tools.large_value({});
|
||||
let failure;
|
||||
try {
|
||||
await tools.always_fail({});
|
||||
} catch (error) {
|
||||
failure = {
|
||||
typed: error instanceof ToolCallError,
|
||||
name: error.name,
|
||||
toolName: error.toolName,
|
||||
message: error.message,
|
||||
exposesCode: 'code' in error,
|
||||
exposesContent: 'content' in error,
|
||||
exposesInfo: 'info' in error,
|
||||
};
|
||||
}
|
||||
return { length: large.length, failure };
|
||||
`))
|
||||
|
||||
expect(value).toEqual({
|
||||
length: 100_000,
|
||||
failure: {
|
||||
typed: true,
|
||||
name: 'ToolCallError',
|
||||
toolName: 'always_fail',
|
||||
message: 'expected failure',
|
||||
exposesCode: false,
|
||||
exposesContent: false,
|
||||
exposesInfo: false,
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
it('returns a background task id, settles the outer run, and polls that id to completion', async () => {
|
||||
workdir = await mkdtemp(join(tmpdir(), 'dsh-code-mode-background-'))
|
||||
ctx = await backgroundCodeModeHarness(workdir)
|
||||
|
||||
const taskId = completion(await runCode(ctx, `
|
||||
const started = await tools.bash({
|
||||
command: "sleep 0.2; printf 'background-complete\\n'",
|
||||
description: 'Run completion marker in background',
|
||||
run_in_background: true,
|
||||
});
|
||||
return started.taskId;
|
||||
`))
|
||||
expect(taskId).toBe('bash-1')
|
||||
|
||||
const polled = completion(await runCode(ctx, `
|
||||
return await tools.task_output({ task_id: ${JSON.stringify(taskId)}, wait: true, timeout_ms: 5000 });
|
||||
`))
|
||||
if (typeof polled !== 'object' || polled === null || Array.isArray(polled)) throw new Error('invalid task_output completion')
|
||||
const taskOutput = polled as Record<string, unknown>
|
||||
expect(taskOutput.text).toContain('background-complete')
|
||||
expect(taskOutput.task).toMatchObject({ id: taskId, kind: 'bash', status: 'completed' })
|
||||
}, 15_000)
|
||||
|
||||
it('pre-abort spawns nothing; post-publication abort leaves task_kill as the cancellation owner', async () => {
|
||||
workdir = await mkdtemp(join(tmpdir(), 'dsh-code-mode-task-cancel-'))
|
||||
ctx = await backgroundCodeModeHarness(workdir)
|
||||
|
||||
const pre = new AbortController()
|
||||
pre.abort('pre-aborted')
|
||||
const preResult = await runCode(ctx, `
|
||||
return await tools.bash({ command: 'sleep 10', description: 'Must never start', run_in_background: true });
|
||||
`, pre.signal)
|
||||
expect(preResult.isError).toBe(true)
|
||||
expect(ctx.tasks.list()).toEqual([])
|
||||
|
||||
const afterPublication = new AbortController()
|
||||
const running = runCode(ctx, `
|
||||
const started = await tools.bash({ command: 'sleep 10', description: 'Wait for explicit task kill', run_in_background: true });
|
||||
console.log(started.taskId);
|
||||
await new Promise(() => {});
|
||||
`, afterPublication.signal)
|
||||
for (let attempt = 0; attempt < 100 && ctx.tasks.list().length === 0; attempt++) {
|
||||
await new Promise(resolve => setTimeout(resolve, 10))
|
||||
}
|
||||
const task = ctx.tasks.list()[0]
|
||||
expect(task).toMatchObject({ id: 'bash-1', status: 'running' })
|
||||
afterPublication.abort('outer-call-cancelled')
|
||||
expect((await running).isError).toBe(true)
|
||||
expect(ctx.tasks.list()[0]).toMatchObject({ id: task!.id, status: 'running' })
|
||||
|
||||
const killed = completion(await runCode(ctx, `
|
||||
return await tools.task_kill({ task_id: ${JSON.stringify(task!.id)}, reason: 'test owns cancellation' });
|
||||
`))
|
||||
expect(killed).toMatchObject({ outcome: 'cancellation-requested', task: { id: task!.id } })
|
||||
const settled = completion(await runCode(ctx, `
|
||||
return await tools.task_output({ task_id: ${JSON.stringify(task!.id)}, wait: true, timeout_ms: 5000 });
|
||||
`))
|
||||
expect(settled).toMatchObject({ task: { id: task!.id, status: 'killed' } })
|
||||
}, 15_000)
|
||||
|
||||
it('keeps foreground bash coupled to the outer signal', async () => {
|
||||
workdir = await mkdtemp(join(tmpdir(), 'dsh-code-mode-foreground-cancel-'))
|
||||
ctx = await backgroundCodeModeHarness(workdir)
|
||||
const controller = new AbortController()
|
||||
const startedAt = Date.now()
|
||||
const pending = runCode(ctx, `
|
||||
return await tools.bash({ command: 'sleep 10', description: 'Run cancellable foreground command' });
|
||||
`, controller.signal)
|
||||
setTimeout(() => { controller.abort('stop-foreground') }, 200)
|
||||
const result = await pending
|
||||
expect(result.isError).toBe(true)
|
||||
expect(Date.now() - startedAt).toBeLessThan(5_000)
|
||||
expect(ctx.tasks.list()).toEqual([])
|
||||
}, 15_000)
|
||||
|
||||
it('uses cordis_mount DTO ids directly for active and pending mounts, then confirms removal', async () => {
|
||||
ctx = await typedCodeModeHarness()
|
||||
await ctx.plugin(ToolCordis)
|
||||
|
||||
const value = completion(await runCode(ctx, `
|
||||
const active = await tools.cordis_mount({
|
||||
code: "return { name: 'active-code-mode-plugin', apply(ctx) {} }",
|
||||
});
|
||||
const pending = await tools.cordis_mount({
|
||||
code: "return { name: 'pending-code-mode-plugin', inject: ['missing-code-mode-service'], apply(ctx) {} }",
|
||||
});
|
||||
const before = await tools.cordis_inspect({ what: 'dynamic' });
|
||||
const unmounted = await tools.cordis_unmount({ id: active.id });
|
||||
const after = await tools.cordis_inspect({ what: 'dynamic' });
|
||||
await tools.cordis_unmount({ id: pending.id });
|
||||
return {
|
||||
active,
|
||||
pending,
|
||||
unmounted,
|
||||
beforeContainsId: before.includes(active.id),
|
||||
afterContainsId: after.includes(active.id),
|
||||
};
|
||||
`))
|
||||
|
||||
expect(value).toEqual({
|
||||
active: {
|
||||
id: 'dyn-1',
|
||||
pluginName: 'active-code-mode-plugin',
|
||||
state: 'active',
|
||||
provides: [],
|
||||
waitingFor: [],
|
||||
},
|
||||
pending: {
|
||||
id: 'dyn-2',
|
||||
pluginName: 'pending-code-mode-plugin',
|
||||
state: 'pending',
|
||||
provides: [],
|
||||
waitingFor: ['missing-code-mode-service'],
|
||||
},
|
||||
unmounted: { id: 'dyn-1', pluginName: 'active-code-mode-plugin' },
|
||||
beforeContainsId: true,
|
||||
afterContainsId: false,
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
function waitForIdle(harness: Context, agent: Agent): Promise<void> {
|
||||
return new Promise((resolve) => {
|
||||
const dispose = harness.on('agent/status', (subject, status) => {
|
||||
|
||||
@@ -92,7 +92,7 @@
|
||||
{"type":"assistant/chunk","seq":90,"time":1783611772685,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"return"}}}
|
||||
{"type":"assistant/chunk","seq":91,"time":1783611772685,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" out"}}}
|
||||
{"type":"assistant/chunk","seq":92,"time":1783611772713,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"1"}}}
|
||||
{"type":"assistant/chunk","seq":93,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".trim"}}}
|
||||
{"type":"assistant/chunk","seq":93,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".stdout.text.trim"}}}
|
||||
{"type":"assistant/chunk","seq":94,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"()"}}}
|
||||
{"type":"assistant/chunk","seq":95,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" +"}}}
|
||||
{"type":"assistant/chunk","seq":96,"time":1783611772714,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" \\\"+"}}}
|
||||
@@ -100,16 +100,16 @@
|
||||
{"type":"assistant/chunk","seq":98,"time":1783611772744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" +"}}}
|
||||
{"type":"assistant/chunk","seq":99,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":" out"}}}
|
||||
{"type":"assistant/chunk","seq":100,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"2"}}}
|
||||
{"type":"assistant/chunk","seq":101,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".trim"}}}
|
||||
{"type":"assistant/chunk","seq":101,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":".stdout.text.trim"}}}
|
||||
{"type":"assistant/chunk","seq":102,"time":1783611772745,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"();"}}}
|
||||
{"type":"assistant/chunk","seq":103,"time":1783611772772,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"\""}}}
|
||||
{"type":"assistant/chunk","seq":104,"time":1783611772773,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","argumentsDelta":"}"}}}
|
||||
{"type":"assistant/chunk","seq":105,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"reasoning","text":"The user wants a single run_code program that calls bash twice, then returns the two outputs joined with a plus sign. Let me write this."}}}}
|
||||
{"type":"assistant/chunk","seq":106,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.trim() + \\\"+\\\" + out2.trim();\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":106,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.stdout.text.trim() + \\\"+\\\" + out2.stdout.text.trim();\"}"}}}}
|
||||
{"type":"assistant/chunk","seq":107,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":3009,"outputTokens":133,"cacheReadTokens":0,"reasoningTokens":29}}}}
|
||||
{"type":"assistant/chunk","seq":108,"time":1783611772836,"data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"tool-calls"}}}}
|
||||
{"type":"assistant/message","seq":109,"time":1783611772840,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants a single run_code program that calls bash twice, then returns the two outputs joined with a plus sign. Let me write this."},{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.trim() + \\\"+\\\" + out2.trim();\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":3009,"outputTokens":133,"cacheReadTokens":0,"reasoningTokens":29}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":110,"time":1783611772840,"data":{"turn":1,"step":1,"callId":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.trim() + \\\"+\\\" + out2.trim();\"}"}}
|
||||
{"type":"assistant/message","seq":109,"time":1783611772840,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants a single run_code program that calls bash twice, then returns the two outputs joined with a plus sign. Let me write this."},{"type":"tool-call","id":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.stdout.text.trim() + \\\"+\\\" + out2.stdout.text.trim();\"}"}],"provenance":{"provider":"deepseek","model":"deepseek-v4-flash"},"usage":{"inputTokens":3009,"outputTokens":133,"cacheReadTokens":0,"reasoningTokens":29}},"sourceEventSeqs":[4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108],"surfaceOp":"append"}
|
||||
{"type":"tool/call","seq":110,"time":1783611772840,"data":{"turn":1,"step":1,"callId":"call_00_DRxnM6R1TThfDwcudW0f2050","name":"run_code","arguments":"{\"code\": \"const out1 = await tools.bash({ command: \\\"echo CODE_ONE\\\", description: \\\"First echo\\\" });\\nconst out2 = await tools.bash({ command: \\\"echo CODE_TWO\\\", description: \\\"Second echo\\\" });\\nreturn out1.stdout.text.trim() + \\\"+\\\" + out2.stdout.text.trim();\"}"}}
|
||||
{"type":"tool/code-dispatch","seq":111,"time":1783611772933,"data":{"parentCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","subCallId":"call_00_DRxnM6R1TThfDwcudW0f2050:code:1","name":"bash","arguments":{"command":"echo CODE_ONE","description":"First echo"},"isError":false,"resultSummary":"CODE_ONE\n"}}
|
||||
{"type":"tool/code-dispatch","seq":112,"time":1783611772936,"data":{"parentCallId":"call_00_DRxnM6R1TThfDwcudW0f2050","subCallId":"call_00_DRxnM6R1TThfDwcudW0f2050:code:2","name":"bash","arguments":{"command":"echo CODE_TWO","description":"Second echo"},"isError":false,"resultSummary":"CODE_TWO\n"}}
|
||||
{"type":"tool/result","seq":113,"time":1783611772937,"data":{"turn":1,"step":1,"callId":"call_00_DRxnM6R1TThfDwcudW0f2050","content":[{"type":"text","text":"CODE_ONE+CODE_TWO"}],"isError":false,"meta":{"logs":[]}},"sourceEventSeqs":[110],"surfaceOp":"append"}
|
||||
|
||||
Reference in New Issue
Block a user