Merge remote-tracking branch 'origin/master' into worktree/web-multimodal-image-input
# Conflicts: # .agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.i18n.yaml # .agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md # .agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.zh.md # .agents/notes/implemented/architecture/2026-07-25-web-input-machine-and-slash-pipeline.i18n.yaml # .agents/notes/implemented/architecture/2026-07-25-web-input-machine-and-slash-pipeline.md # .agents/notes/implemented/architecture/2026-07-25-web-input-machine-and-slash-pipeline.zh.md # THIRD_PARTY_NOTICES.md # apps/cli/composition.md # apps/cli/config/base.cordis.yml # apps/cli/package.json # apps/cli/src/app-cli-entry.ts # apps/cli/src/bin.ts # apps/cli/tests/args.spec.ts # apps/web/tests/built-boot.snapshot.ts # apps/web/tests/navigation-panes.e2e.ts # docs/architecture.i18n.yaml # docs/architecture.md # docs/architecture.zh.md # docs/config-catalog.md # docs/cordis-catalog/services.md # docs/core-data-structures/core.i18n.yaml # docs/core-data-structures/llm-streaming.i18n.yaml # docs/event-producer-consumer.md # docs/module-graph.md # examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/session.jsonl # packages/README.i18n.yaml # packages/bundle/README.i18n.yaml # packages/client/connection/README.i18n.yaml # packages/client/connection/README.md # packages/client/connection/README.zh.md # packages/client/connection/src/client/fixture.ts # packages/client/connection/src/http-bridge.ts # packages/client/connection/src/index.ts # packages/client/connection/tests/fixture.spec.ts # packages/client/connection/tests/node-half.spec.ts # packages/client/runtime/README.i18n.yaml # packages/client/runtime/README.md # packages/client/runtime/README.zh.md # packages/client/runtime/src/client/contract/session.ts # packages/client/runtime/src/client/sessions/session.ts # packages/client/ui-conversation/README.i18n.yaml # packages/client/ui-conversation/README.md # packages/client/ui-conversation/README.zh.md # packages/client/ui-conversation/src/client/apply.ts # packages/client/ui-conversation/src/client/chat/AssistantMarkdown.tsx # packages/client/ui-conversation/src/client/chat/ChatView.tsx # packages/client/ui-conversation/src/client/chat/MessageItem.module.css # packages/client/ui-conversation/src/client/chat/MessageItem.tsx # packages/client/ui-conversation/src/client/contract/slots.ts # packages/client/ui-conversation/src/client/index.ts # packages/client/ui-conversation/src/client/input/contract.ts # packages/client/ui-conversation/src/client/input/facade.ts # packages/client/ui-conversation/src/client/input/hub.ts # packages/client/ui-conversation/src/client/locales.ts # packages/client/ui-conversation/src/client/service.ts # packages/client/ui-conversation/src/client/skeleton/ConversationSession.tsx # packages/client/ui-conversation/src/client/skeleton/InputBar.tsx # packages/client/ui-conversation/tests/apply-inject.spec.tsx # packages/client/ui-conversation/tests/input-bar.spec.tsx # packages/client/ui-conversation/tests/input-matrix.spec.tsx # packages/client/ui-conversation/tests/input-scenarios.spec.tsx # packages/client/ui-conversation/tests/service-orchestration.spec.ts # packages/client/ui-conversation/tests/skeleton.spec.tsx # packages/client/ui-trajectory/tests/views.spec.tsx # packages/compact/compact-basic/README.i18n.yaml # packages/cordis/tool-cordis/src/api-catalog.ts # packages/host/apiproxy/README.i18n.yaml # packages/host/apiproxy/README.md # packages/host/apiproxy/README.zh.md # packages/host/apiproxy/src/api-proxy.ts # packages/host/apiproxy/src/api/rpc.ts # packages/host/apiproxy/src/api/sessions.ts # packages/host/apiproxy/src/index.ts # packages/host/apiproxy/tests/api-proxy-models.spec.ts # packages/host/apiproxy/tests/rpc-schemas.spec.ts # packages/llm/llm-pi-ai/README.i18n.yaml # packages/llm/llm-pi-ai/README.md # packages/llm/llm-pi-ai/README.zh.md # packages/llm/llm-pi-ai/src/adapter.ts # packages/llm/llm/README.i18n.yaml # packages/ui/tui/README.md # packages/ui/tui/README.zh.md # packages/ui/tui/src/components/content.ts # packages/ui/tui/src/components/transcript.ts # packages/ui/tui/tests/tui.spec.ts # pnpm-lock.yaml
This commit is contained in:
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/README.md
|
||||
README.md: 66b7beabd73cc3fec7230f209a9da0da48a37c95
|
||||
README.zh.md: 3f38c94fb4a43bed007d06b8c8558b5bd330a41c
|
||||
README.md: 92d9fbfa2b8c8db4700562009db49229b2189ab3
|
||||
README.zh.md: 5c6e7aad1db6511bdb660b86e257652128db131f
|
||||
|
||||
@@ -6,10 +6,10 @@ The LLM seam and its provider adapters. The interface package (`llm`) owns the a
|
||||
|
||||
| Package | Role | ctx key |
|
||||
|---|---|---|
|
||||
| `llm/` | Abstract LLM service + content-block vocabulary + chunk assembler | `ctx.llm` |
|
||||
| `token-meter/` | Replay-aware request and surface token measurement | `ctx.tokenMeter` |
|
||||
| `llm-retry/` | Exact-provider normal or unbounded request retry policy | (listens to `agent/request-error`) |
|
||||
| `llm-deepseek/` | DeepSeek API adapter (direct fetch + eventsource-parser SSE) | (registers on `ctx.llm`) |
|
||||
| `llm-pi-ai/` | Multi-provider adapter via `@earendil-works/pi-ai` | (registers on `ctx.llm`) |
|
||||
| [`llm/`](llm/README.md) | LLM service and shared streaming vocabulary | `ctx.llm` |
|
||||
| [`token-meter/`](token-meter/README.md) | Replay-aware token measurement | `ctx.tokenMeter` |
|
||||
| [`llm-retry/`](llm-retry/README.md) | Provider-scoped retry policy | listens to `agent/request-error` |
|
||||
| [`llm-deepseek/`](llm-deepseek/README.md) | Direct DeepSeek adapter | registers on `ctx.llm` |
|
||||
| [`llm-pi-ai/`](llm-pi-ai/README.md) | Multi-provider pi-ai adapter | registers on `ctx.llm` |
|
||||
|
||||
The interface lives at `llm/llm/`; adapters, retry policy, and the reusable token meter are flat siblings under the group. Requests route by `provider`, while `model` is passed through to the selected adapter. The route-owning adapter supplies retry policy and resolves available exact-model identity, context capacity, and reasoning metadata; the retry executor and token meter remain provider-agnostic. A new provider adapter registers one or more provider routes on `ctx.llm` without touching the consumers. See [twin LLM adapters](../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md) for the two shipping implementations, the [replay token meter Agent Note](../../.agents/notes/implemented/architecture/2026-07-15-replay-token-meter-service.md) for measurement ownership, and the [routed model context Agent Note](../../.agents/notes/implemented/architecture/2026-07-20-routed-model-context-and-compaction-policy.md) for capacity and compaction-policy ownership.
|
||||
Adapters register provider routes on the seam; retry and token measurement remain separate consumers. The child READMEs own routing, metadata, replay, and provider-wire details; the [LLM architecture decisions](../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md) own the rationale.
|
||||
|
||||
@@ -1,15 +1,15 @@
|
||||
# llm/:LLM(大语言模型)能力家族
|
||||
# llm/ — LLM 能力家族
|
||||
|
||||
[English](README.md) | 中文
|
||||
|
||||
LLM seam 及其提供方适配器。接口包(`llm`)拥有抽象服务、内容块词汇和流分片组装器;适配器是在 `ctx.llm` 上注册的具体实现。这些全是**产品**包(package)。
|
||||
LLM(大语言模型)seam 及其提供方适配器。接口包(`llm`)负责抽象服务、内容块词汇和流式分片组装器;适配器是注册到 `ctx.llm` 的具体实现。这些全是**产品**包。
|
||||
|
||||
| 包 | 职责 | ctx key |
|
||||
|---|---|---|
|
||||
| `llm/` | 抽象 LLM 服务 + 内容块词汇 + 分片组装器 | `ctx.llm` |
|
||||
| `token-meter/` | 感知回放的请求 token 与表层 token 测量 | `ctx.tokenMeter` |
|
||||
| `llm-retry/` | 确切提供方的常规或无界请求重试策略 | (监听 `agent/request-error`) |
|
||||
| `llm-deepseek/` | DeepSeek API 适配器,直接使用 fetch + eventsource-parser 和 SSE(Server-Sent Events) | (注册到 `ctx.llm`) |
|
||||
| `llm-pi-ai/` | 通过 `@earendil-works/pi-ai` 实现的多提供方适配器 | (注册到 `ctx.llm`) |
|
||||
| [`llm/`](llm/README.md) | LLM 服务和共享流式词汇 | `ctx.llm` |
|
||||
| [`token-meter/`](token-meter/README.md) | 可感知回放的 token 测量 | `ctx.tokenMeter` |
|
||||
| [`llm-retry/`](llm-retry/README.md) | 提供方作用域的重试策略 | 监听 `agent/request-error` |
|
||||
| [`llm-deepseek/`](llm-deepseek/README.md) | 直接 DeepSeek 适配器 | 注册到 `ctx.llm` |
|
||||
| [`llm-pi-ai/`](llm-pi-ai/README.md) | 多提供方 pi-ai 适配器 | 注册到 `ctx.llm` |
|
||||
|
||||
接口位于 `llm/llm/`;适配器、重试策略和可复用的 token 计量器以扁平结构并列在该分组下。请求按 `provider` 路由,而 `model` 会原样传给选中的适配器。负责该路由的适配器提供重试策略,并解析可用的确切模型身份、上下文容量和推理元数据;重试执行器与 token 计量器仍与提供方无关。新的提供方适配器只需在 `ctx.llm` 上注册一个或多个提供方路由,无需改动消费方。两个已交付实现见[双生 LLM 适配器](../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md),测量归属见[回放 token 计量器 Agent Note(agent 决策记录)](../../.agents/notes/implemented/architecture/2026-07-15-replay-token-meter-service.md),容量与压缩(compaction)策略归属见[路由模型上下文 Agent Note](../../.agents/notes/implemented/architecture/2026-07-20-routed-model-context-and-compaction-policy.md)。
|
||||
适配器在 seam 上注册提供方路由;重试与 token 测量仍是独立消费方。子 README 负责路由、元数据、回放和提供方协议细节;[LLM 架构决策](../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md)负责设计原理。
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/llm-deepseek/README.md
|
||||
README.md: 020aa65073495526be3f32912b7cd06667c52a2e
|
||||
README.zh.md: 4c655e90ba00340c056f6ac16159621f7a8c1ddb
|
||||
README.md: c6435d0bdfbb9758b6f86ef38e94159a9ccdc36d
|
||||
README.zh.md: f37286ade023ceecf6bfc87eb08fd4d80a5a3912
|
||||
|
||||
@@ -40,7 +40,7 @@ The plugin registers the single provider route `deepseek-official` together with
|
||||
|
||||
`contextWindow` is optional per configured model and is not exposed through the advisory catalog. `ctx.llm.resolveModelInfo('deepseek-official', model).context` returns an exact model value first, then `defaultContextWindow` for an entry without capacity or an unlisted pass-through id. The adapter default is 1,000,000; pressure-sensitive plugins therefore get deployment-owned capacity without treating the model selector as authoritative. Registering another adapter for `deepseek-official` throws `LlmError('DUPLICATE_ADAPTER')`.
|
||||
|
||||
`maxTokens` is the adapter-configured output cap for conversation requests and defaults to 256,000. Exact-model resolution exposes it as `defaultMaxTokens`; `LlmService` materializes that value into `GenerateOptions.maxTokens` before the agent loop writes `request/header`, so the wire request remains reconstructable. An explicit request or `AgentOptions.maxTokens` value wins and is serialized as `max_tokens`. The adapter does not clamp this request budget against `contextWindow`; deployments with a smaller context or provider output limit must configure a compatible `maxTokens`.
|
||||
`maxTokens` is the adapter-configured output cap for conversation requests and defaults to 256,000. A catalog entry may carry its own `maxTokens`, which wins for that model; an entry without one, and any unlisted pass-through id, resolve to the profile value, so adding a per-model cap changes one model rather than the route. Exact-model resolution exposes the winner as `defaultMaxTokens`; `LlmService` materializes that value into `GenerateOptions.maxTokens` before the agent loop writes `request/header`, so the wire request remains reconstructable. An explicit request or `AgentOptions.maxTokens` value wins and is serialized as `max_tokens`. The adapter does not clamp this request budget against `contextWindow`; deployments with a smaller context or provider output limit must configure a compatible `maxTokens`.
|
||||
|
||||
The same exact-model result exposes ordered `off`, `high`, and `max` efforts under `reasoning` for every pass-through model when deployment policy permits thinking. `reasoningEffort` selects the deployment default and falls back to `high` when omitted. `agent/request` can replace it on each conversation step; the resolved value is logged in `request/header`. `high` and `max` enable thinking and serialize as the official top-level `reasoning_effort`; adapter-owned `off` instead serializes `thinking.type: disabled` and omits `reasoning_effort`. An unsupported value fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O.
|
||||
|
||||
@@ -53,7 +53,7 @@ The same exact-model result exposes ordered `off`, `high`, and `max` efforts und
|
||||
Connection facts are not frozen at load. `resolveAdapterOptions` is the one explicit resolve step from raw config to validated facts, and the adapter re-reads them through a thunk **once per operation**: base URL, catalog, request defaults, and idle budget all take effect on the next request, while an in-flight stream keeps the facts it started with. Two optional seams feed that thunk:
|
||||
|
||||
- **`ctx.settings`** — the plugin registers the `llm-deepseek` namespace with this same `Config` schema and its `cordis.yml` entry as the composition `base`, so a `llm-deepseek:` section in the user settings document overrides any field without a restart. Without a mounted settings service the entry config alone drives the adapter, unchanged. A live settings snapshot that passes the schema but fails a beyond-schema bound (a duplicate catalog id, a broken thinking/effort pair) keeps the last good facts and logs the failure; the entry config itself still fails plugin load.
|
||||
- **`ctx.credentials`** — the API key resolves per stream call, from the *same* resolved snapshot that supplies the endpoint: a non-empty literal `apiKey` wins, then `apiKeyEnv` through the credential seam (`$DSH_HOME/.env` under the live environment), then — only without a mounted seam — the raw environment variable. Because credential facts travel with the connection facts, a settings snapshot the resolver rejects contributes neither its endpoint nor its key: the whole previous generation keeps serving. A request with no key anywhere fails with `MISSING_CREDENTIAL` naming every configuration entry point, while the route stays registered and the catalog stays browsable — first-run onboarding is "browse models, store the key, prompt again", with no restart between.
|
||||
- **`ctx.credentials`** — the API key resolves per stream call, from the *same* resolved snapshot that supplies the endpoint: a trimmed, non-empty literal `apiKey` wins, then `apiKeyEnv` through the credential seam (`$DSH_HOME/.env` under the live environment), then — only without a mounted seam — the raw environment variable. Whitespace-only literals are absent rather than Authorization values. Because credential facts travel with the connection facts, a settings snapshot the resolver rejects contributes neither its endpoint nor its key: the whole previous generation keeps serving. Every key is format-checked before use — a literal at connection-facts resolution (plugin load, or the next settings snapshot), a stored or ambient value at request time — so a value no HTTP header can carry is refused there instead of surfacing as an opaque `fetch` `TypeError`; the request-time check throws `LlmError('INVALID_CREDENTIAL')` naming the failing entry point but never any part of the key. A request with no key anywhere fails with `MISSING_CREDENTIAL` naming every configuration entry point, while the route stays registered and the catalog stays browsable — first-run onboarding is "browse models, store the key, prompt again", with no restart between.
|
||||
|
||||
The one registration-captured fact is the retry policy: when its resolved value changes, the plugin re-registers the route in place (same adapter instance, one synchronous section), so `ctx.llm.providerRetryPolicy('deepseek-official')` always reports the current policy.
|
||||
|
||||
@@ -63,7 +63,7 @@ The plugin also declares its route in the configurable-provider directory (`ctx.
|
||||
|
||||
Every request carries the shared attribution header from dsh-llm's `attributionHeaders()` - the mandatory `User-Agent` baseline identifying the harness (see [dsh-llm § App attribution](../llm/README.md#app-attribution-attributionts)). Direct DeepSeek requests and OpenAI-compatible gateway requests get no provider-specific app-attribution headers under this adapter contract; OpenRouter app attribution is deferred to a future explicit OpenRouter adapter or mode. A request whose `GenerateOptions.purpose` is `compaction` (dsh-compact-basic's auxiliary summarization call) additionally carries `x-deepseek-harness-compact: 1`, so the host can separate compaction traffic from conversation requests.
|
||||
|
||||
## Wire-format notes (verified live + against the official docs)
|
||||
## Wire-format notes
|
||||
|
||||
- Streaming only (`stream_options.include_usage` always on). `usage` may arrive attached to the finish chunk or as a trailing usage-only chunk — the translator defers both to `[DONE]`, so `usage` always precedes `finish` and nothing follows `finish`.
|
||||
- The adapter-owned `off` effort maps to `thinking: {type: 'disabled'}` and never crosses the wire as `reasoning_effort: 'off'`.
|
||||
@@ -75,10 +75,6 @@ Every request carries the shared attribution header from dsh-llm's `attributionH
|
||||
|
||||
Non-2xx responses throw `LlmError` with stable codes: `AUTH` (401/403), `QUOTA` (a response whose provider details identify exhausted quota, balance, or credits), `RATE_LIMIT` (other 429s), `CONTEXT_WINDOW_EXCEEDED` (a 400 whose provider code, type, or message identifies context overflow), `INVALID_REQUEST` (other 400s), `SERVER` (5xx), `HTTP_<status>` otherwise. Its serializable `failure` retains the HTTP status plus a valid positive `Retry-After` seconds/date delay and `x-request-id` / `x-deepseek-request-id` when present. A pre-response transport failure (DNS, refused connection, TLS, proxy) throws `TRANSPORT` naming the configured endpoint and chaining the original rejection as `cause`; caller aborts throw `ABORTED`, and the loop's cancellation signal remains authoritative. Protocol violations throw `STREAM_CLOSED` (no `[DONE]`) or `MALFORMED_RESPONSE` (bad JSON payload). Unknown wire `finish_reason`s (e.g. `content_filter`, `insufficient_system_resource`) become `finish {kind: 'error', failure}` chunks, and a completed stream whose `stop` (or absent) finish opened no content blocks becomes a `finish {kind: 'error'}` with code `EMPTY_RESPONSE` (retried by default policy).
|
||||
|
||||
## Testing
|
||||
|
||||
Unit suites run against a local `node:http` mock SSE server (no network), including dynamic `high`/`off`/`max` selection, structured HTTP facts, malformed/truncated streams, caller abort, connection failure, and proof that idle timeout aborts the actual body. `tests/dynamic-config.spec.ts` drives real settings-local and credentials-local providers (next-request base-URL/key pickup, literal precedence, keyless onboarding, last-good snapshots, retry-policy re-registration), and `tests/loader-composition.spec.ts` boots the full chain from a test-only `cordis.yml` through the actual Loader and edits `settings.yaml`/`.env` on disk. Real-API coverage lives in `tests/adapter.e2e.ts` (`pnpm run test:e2e`, key-gated): V4 Flash + V4 Pro across thinking enabled/disabled and both official effort levels, including the thinking+tools round trip with reasoning passback and a request whose key exists only in a credentials-local document.
|
||||
|
||||
## Model Experience
|
||||
|
||||
### DeepSeek request
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
|
||||
harness LLM(大语言模型)seam 的 DeepSeek chat-completions 适配器:直接 `fetch` + SSE(Server-Sent Events,由 `eventsource-parser` 分帧),将官方协议格式(wire format;真源:API 文档 guides/thinking_mode、guides/tool_calls、api/create-chat-completion)转换为 `StreamChunk` 协议。
|
||||
|
||||
同一 seam 的第二个基于库的实现位于 `@deepseek-ai/dsh-llm-pi-ai`。本包(package)拥有 `deepseek-official` 提供方路由——刻意区别于 pi-ai 的 catalog 名称 `deepseek`,因此同一组合可以并排挂载两条 DeepSeek 路径;而为 `deepseek-official` 本身注册另一个适配器仍会抛出 `LlmError('DUPLICATE_ADAPTER')`。
|
||||
同一 seam 的第二个基于库的实现位于 `@deepseek-ai/dsh-llm-pi-ai`。本包拥有 `deepseek-official` 提供方路由——刻意区别于 pi-ai 的 catalog 名称 `deepseek`,因此同一组合可以并排挂载两条 DeepSeek 路径;而为 `deepseek-official` 本身注册另一个适配器仍会抛出 `LlmError('DUPLICATE_ADAPTER')`。
|
||||
|
||||
包根入口导出 Cordis 插件契约与 `DeepSeekAdapter`;协议序列化、SSE 解析与分片转换 helper 不属于该根契约。
|
||||
|
||||
@@ -40,7 +40,7 @@ harness LLM(大语言模型)seam 的 DeepSeek chat-completions 适配器:
|
||||
|
||||
`contextWindow` 对每个已配置模型都可选,不会通过建议 catalog 公开。`ctx.llm.resolveModelInfo('deepseek-official', model).context` 先返回精确模型值,再对不含容量的配置项或未列出原样传递 id 返回 `defaultContextWindow`。适配器默认值为 1,000,000;因此,压力敏感插件可以获得由部署决定的容量,不会将模型 selector 视为权威。为 `deepseek-official` 注册另一个适配器会抛出 `LlmError('DUPLICATE_ADAPTER')`。
|
||||
|
||||
`maxTokens` 是适配器为对话请求配置的输出上限,默认值为 256,000。确切模型解析会将其公开为 `defaultMaxTokens`;`LlmService` 会在 agent loop(智能体循环)写入 `request/header` 前,将该值填入 `GenerateOptions.maxTokens`,从而仍可根据持久记录重建协议请求。显式的请求值或 `AgentOptions.maxTokens` 值优先,并会序列化为 `max_tokens`。适配器不会根据 `contextWindow` 自动调低该请求预算;上下文或提供方输出上限较小的部署必须配置与其相容的 `maxTokens`。
|
||||
`maxTokens` 是适配器为对话请求配置的输出上限,默认值为 256,000。Catalog 配置项可以自带 `maxTokens`,它对该模型胜出;不含该上限的配置项以及任何未列出原样传递 id 都解析为 profile 值,因此新增按模型的上限只改变一个模型,而非整条路由。确切模型解析会将胜出值公开为 `defaultMaxTokens`;`LlmService` 会在 agent loop(智能体循环)写入 `request/header` 前,将该值填入 `GenerateOptions.maxTokens`,从而仍可根据持久记录重建协议请求。显式的请求值或 `AgentOptions.maxTokens` 值优先,并会序列化为 `max_tokens`。适配器不会根据 `contextWindow` 自动调低该请求预算;上下文或提供方输出上限较小的部署必须配置与其相容的 `maxTokens`。
|
||||
|
||||
同一确切模型结果会在部署策略允许思考时,为每个原样传递模型在 `reasoning` 下公开有序的 `off`、`high` 和 `max` 推理(reasoning)强度。`reasoningEffort` 选择部署默认值,省略时回退为 `high`。`agent/request` 可以在每个会话步骤替换它;解析后的值会记录在 `request/header`。`high` 和 `max` 会启用思考,并序列化为官方顶层 `reasoning_effort`;适配器持有的 `off` 则序列化为 `thinking.type: disabled`,且省略 `reasoning_effort`。不支持的值会在网络 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败。
|
||||
|
||||
@@ -53,7 +53,7 @@ harness LLM(大语言模型)seam 的 DeepSeek chat-completions 适配器:
|
||||
连接事实不在加载时冻结。`resolveAdapterOptions` 是从原始配置到已校验事实的唯一显式 resolve 步骤,适配器经由一个 thunk **每操作重读一次**:base URL、catalog、请求默认值与 idle 预算都在下一次请求生效,进行中的流则保持其起始事实。两个可选 seam 供给该 thunk:
|
||||
|
||||
- **`ctx.settings`**——插件用同一份 `Config` schema 注册 `llm-deepseek` namespace,并以其 `cordis.yml` 条目为组合 `base`,因此用户设置文档中的 `llm-deepseek:` 分节可以免重启覆盖任何字段。未挂载 settings 服务时,仅由 entry 配置驱动适配器,行为不变。存活 settings 快照若通过 schema 却违反 schema 之外的约束(重复的 catalog id、无法成立的 thinking/推理强度组合),则保留最后可用事实并记录失败;entry 配置本身仍会使插件加载失败。
|
||||
- **`ctx.credentials`**——API 密钥按每次 stream 调用解析,取自与端点*同一*份解析后的快照:非空的字面 `apiKey` 优先,其次经凭据 seam 解析 `apiKeyEnv`(活跃环境之下的 `$DSH_HOME/.env`),最后——仅在未挂载 seam 时——读取原始环境变量。由于凭据事实与连接事实同行,被 resolver 拒绝的 settings 快照既不贡献自己的端点,也不贡献自己的密钥:整个先前世代继续服务。任何地方都没有密钥的请求以 `MISSING_CREDENTIAL` 失败,并点名每个配置入口,同时路由保持注册、catalog 保持可浏览——首次运行的上手流程就是「浏览模型、存入密钥、再次发起提示」,中间无需任何重启。
|
||||
- **`ctx.credentials`**——API 密钥按每次 stream 调用解析,取自与端点*同一*份解析后的快照:去除首尾空白后非空的字面 `apiKey` 优先,其次经凭据 seam 解析 `apiKeyEnv`(活跃环境之下的 `$DSH_HOME/.env`),最后——仅在未挂载 seam 时——读取原始环境变量。纯空白字面值会被视为缺失,而不会成为 Authorization 值。由于凭据事实与连接事实同行,被 resolver 拒绝的 settings 快照既不贡献自己的端点,也不贡献自己的密钥:整个先前世代继续服务。每个密钥在使用前都会被校验格式——字面量在连接事实解析时(插件加载或下一次 settings 快照)校验,已存储的值或环境变量值则在请求时校验——因此 HTTP 标头无法承载的值会在这一步被拒绝,而不是以语义不明的 `fetch` `TypeError` 形式浮现;请求时校验会抛出 `LlmError('INVALID_CREDENTIAL')`,点名失败的入口,但绝不透露密钥的任何部分。任何地方都没有密钥的请求以 `MISSING_CREDENTIAL` 失败,并点名每个配置入口,同时路由保持注册、catalog 保持可浏览——首次运行的上手流程就是「浏览模型、存入密钥、再次发起提示」,中间无需任何重启。
|
||||
|
||||
唯一在注册期捕获的事实是重试策略:其解析值变化时,插件原地重新注册该路由(同一适配器实例、一个同步区段),因此 `ctx.llm.providerRetryPolicy('deepseek-official')` 始终报告当前策略。
|
||||
|
||||
@@ -63,7 +63,7 @@ harness LLM(大语言模型)seam 的 DeepSeek chat-completions 适配器:
|
||||
|
||||
每个请求都携带 dsh-llm `attributionHeaders()` 的共享归因标头,即用于识别 harness 的必需 `User-Agent` 基线(见 [dsh-llm § 应用归因](../llm/README.md#app-attribution-attributionts))。在该适配器契约(adapter contract)下,直接 DeepSeek 请求与 OpenAI 兼容 gateway 请求都不会获得提供方特定应用归因标头;OpenRouter 应用归因暂缓到未来的显式 OpenRouter 适配器或模式。`GenerateOptions.purpose` 为 `compaction` 的请求(dsh-compact-basic 的辅助摘要调用)还会携带 `x-deepseek-harness-compact: 1`,让宿主可以将压缩流量与会话请求分开。
|
||||
|
||||
## 协议格式说明(已通过实时请求与官方文档验证)
|
||||
## 协议格式说明
|
||||
|
||||
- 只支持流式输出(`stream_options.include_usage` 始终开启)。`usage` 可能附着在 finish 分片上,也可能作为尾随的纯 usage 分片到达;转换器会将两者都延迟到 `[DONE]`,因此 `usage` 始终位于 `finish` 之前,`finish` 之后不会出现任何内容。
|
||||
- 适配器持有的 `off` 推理强度映射为 `thinking: {type: 'disabled'}`,绝不会以 `reasoning_effort: 'off'` 通过协议发送。
|
||||
@@ -75,10 +75,6 @@ harness LLM(大语言模型)seam 的 DeepSeek chat-completions 适配器:
|
||||
|
||||
非 2xx 响应会抛出稳定 code 的 `LlmError`:`AUTH`(401/403)、`QUOTA`(提供方详细信息标识配额、余额或点数耗尽的响应)、`RATE_LIMIT`(其他 429)、`CONTEXT_WINDOW_EXCEEDED`(提供方 code、type 或 message 标识上下文溢出的 400)、`INVALID_REQUEST`(其他 400)、`SERVER`(5xx),其他情况为 `HTTP_<status>`。其可序列化 `failure` 保留 HTTP 状态,以及有效的正 `Retry-After` 秒数/日期延迟和存在时的 `x-request-id` / `x-deepseek-request-id`。响应前传输失败(DNS、连接被拒绝、TLS、proxy)会抛出命名已配置端点的 `TRANSPORT`,并将原始拒绝作为 `cause`;调用方 abort 抛出 `ABORTED`,仍以 loop 的取消信号为准。协议违例抛出 `STREAM_CLOSED`(没有 `[DONE]`)或 `MALFORMED_RESPONSE`(JSON payload 格式错误)。未知协议 `finish_reason`(例如 `content_filter`、`insufficient_system_resource`)会变为 `finish {kind: 'error', failure}` 分片;已完成流如果使用 `stop`(或缺失)finish 但没有开启内容块,就会变为 `finish {kind: 'error'}`,code 为 `EMPTY_RESPONSE`(默认策略会重试)。
|
||||
|
||||
## 测试
|
||||
|
||||
单元套件使用本地 `node:http` mock SSE 服务器(无网络),覆盖动态 `high`/`off`/`max` 选择、结构化 HTTP 事实、格式错误/截断流、调用方 abort、连接失败,以及 idle 超时确实会 abort 实际 body 的证明。`tests/dynamic-config.spec.ts` 驱动真实的 settings-local 与 credentials-local provider(下一请求即生效的 base-URL/密钥拾取、字面值优先、无密钥上手、最后可用快照、重试策略重注册),`tests/loader-composition.spec.ts` 则从仅测试用的 `cordis.yml` 出发,经真实 Loader 拉起完整链路,并在磁盘上编辑 `settings.yaml`/`.env`。真实 API 覆盖位于 `tests/adapter.e2e.ts`(`pnpm run test:e2e`,需有 key 才会运行):V4 Flash + V4 Pro,覆盖思考启用/禁用与两种官方 effort 级别,包括思考 + 工具往返与推理回传,以及密钥仅存在于 credentials-local 文档中的请求。
|
||||
|
||||
## 模型体验
|
||||
|
||||
### DeepSeek 请求
|
||||
|
||||
@@ -21,9 +21,7 @@
|
||||
"files": [
|
||||
"lib/index.js",
|
||||
"lib/invariant.js",
|
||||
"lib/types/**/*.d.ts",
|
||||
"lib/types/**/*.d.ts.map",
|
||||
"src"
|
||||
"lib/types/**/*.d.ts"
|
||||
],
|
||||
"license": "BSD-3-Clause",
|
||||
"peerDependencies": {
|
||||
|
||||
@@ -35,6 +35,8 @@ export interface DeepSeekCatalogModel {
|
||||
description?: string
|
||||
/** Known combined request/response context capacity; omitted when deployment metadata is unavailable. */
|
||||
contextWindow?: number
|
||||
/** Per-request output cap for this model; omission falls back to the profile's {@link DeepSeekConnectionOptions.maxTokens}. */
|
||||
maxTokens?: number
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -186,7 +188,7 @@ export class DeepSeekAdapter extends LlmAdapter {
|
||||
? { provider, id: model, name: model, inputModalities: ['text' as const] }
|
||||
: modelInfo(provider, configured),
|
||||
context: { contextWindow },
|
||||
defaultMaxTokens: connection.maxTokens,
|
||||
defaultMaxTokens: configured?.maxTokens ?? connection.maxTokens,
|
||||
...connection.defaults.thinking === 'disabled'
|
||||
? {
|
||||
reasoning: {
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
|
||||
import type { Context } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import { LlmError, resolveRetryPolicy, RetryPolicySchema } from '@deepseek-ai/dsh-llm'
|
||||
import { assertUsableApiKey, LlmError, normalizeApiKey, resolveRetryPolicy, RetryPolicySchema } from '@deepseek-ai/dsh-llm'
|
||||
import type { RetryPolicyConfig } from '@deepseek-ai/dsh-llm'
|
||||
import { credentialRef } from '@deepseek-ai/dsh-credentials'
|
||||
import { deepEqualJson, installSettingsSection, settingsNamespace } from '@deepseek-ai/dsh-settings'
|
||||
@@ -58,7 +58,13 @@ const DEFAULT_MODELS: DeepSeekCatalogModel[] = [
|
||||
* reasoning effort resolves to `high`.
|
||||
*/
|
||||
export interface Config {
|
||||
/** Literal API key; prefer {@link apiKeyEnv} so no secret enters configuration files. */
|
||||
/**
|
||||
* Trimmed literal API key; whitespace-only is absent, so it resolves through
|
||||
* {@link apiKeyEnv} like an omitted one. Prefer {@link apiKeyEnv} to keep
|
||||
* secrets out of configuration files. {@link resolveAdapterOptions} also
|
||||
* format-checks what remains: a value no HTTP header can carry fails there
|
||||
* rather than inside `fetch`.
|
||||
*/
|
||||
apiKey?: string
|
||||
/** Credential reference (environment-variable name) resolved per request; defaults to `DEEPSEEK_API_KEY`. */
|
||||
apiKeyEnv?: string
|
||||
@@ -68,7 +74,7 @@ export interface Config {
|
||||
thinking?: 'enabled' | 'disabled'
|
||||
/** Default thinking effort (default `high`); `off` disables thinking per request. */
|
||||
reasoningEffort?: 'off' | 'high' | 'max'
|
||||
/** Default per-request output cap (default 256,000); explicit request values win. */
|
||||
/** Default per-request output cap (default 256,000); a model's own cap and explicit request values win. */
|
||||
maxTokens?: number
|
||||
/** Positive context capacity used when the selected model has no exact value (default 1,000,000). */
|
||||
defaultContextWindow?: number
|
||||
@@ -85,6 +91,7 @@ const catalogModel: z<DeepSeekCatalogModel> = z.object({
|
||||
name: z.string(),
|
||||
description: z.string(),
|
||||
contextWindow: z.number().step(1).min(1),
|
||||
maxTokens: z.number().step(1).min(1),
|
||||
})
|
||||
|
||||
export const Config: z<Config> = z.object({
|
||||
@@ -125,6 +132,12 @@ function resolveModels(models: readonly DeepSeekCatalogModel[] | undefined): Dee
|
||||
`llm-deepseek: catalog model "${model.id}" contextWindow must be a positive integer`,
|
||||
)
|
||||
}
|
||||
if (model.maxTokens !== undefined
|
||||
&& (!Number.isInteger(model.maxTokens) || model.maxTokens <= 0)) {
|
||||
throw new Error(
|
||||
`llm-deepseek: catalog model "${model.id}" maxTokens must be a positive integer`,
|
||||
)
|
||||
}
|
||||
if (seen.has(model.id)) throw new Error(`llm-deepseek: duplicate catalog model "${model.id}"`)
|
||||
seen.add(model.id)
|
||||
return {
|
||||
@@ -132,6 +145,7 @@ function resolveModels(models: readonly DeepSeekCatalogModel[] | undefined): Dee
|
||||
...model.name === undefined ? {} : { name: model.name },
|
||||
...model.description === undefined ? {} : { description: model.description },
|
||||
...model.contextWindow === undefined ? {} : { contextWindow: model.contextWindow },
|
||||
...model.maxTokens === undefined ? {} : { maxTokens: model.maxTokens },
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -166,8 +180,25 @@ export function resolveAdapterOptions(config: Config): ResolvedDeepSeekOptions {
|
||||
`llm-deepseek: streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`,
|
||||
)
|
||||
}
|
||||
// An absent apiKey is not a failure: it falls through to apiKeyEnv below.
|
||||
// A supplied one must be usable, so a malformed literal fails here beside
|
||||
// the other beyond-schema bounds instead of inside `fetch`.
|
||||
// Absence is not a failure, and a blank literal is absence: both resolve
|
||||
// through apiKeyEnv below, which is this adapter's defined fallback. (The
|
||||
// pi-ai adapter refuses a blank one instead, because there absence selects a
|
||||
// different authentication mode rather than a different source for the same
|
||||
// key.) What a literal cannot be is unusable: a value no HTTP header can
|
||||
// carry fails here beside the other beyond-schema bounds, not inside `fetch`.
|
||||
let apiKey: string | undefined
|
||||
if (config.apiKey !== undefined) {
|
||||
const checked = normalizeApiKey(config.apiKey)
|
||||
if (!checked.ok && checked.reason === 'illegalCharacters') {
|
||||
throw new Error('llm-deepseek: apiKey contains characters no HTTP header can carry; paste the raw key only')
|
||||
}
|
||||
apiKey = checked.ok ? checked.value : undefined
|
||||
}
|
||||
return {
|
||||
...config.apiKey !== undefined && config.apiKey.length > 0 ? { apiKey: config.apiKey } : {},
|
||||
...apiKey === undefined ? {} : { apiKey },
|
||||
apiKeyEnv: credentialRef(config.apiKeyEnv ?? DEFAULT_API_KEY_ENV),
|
||||
baseURL: config.baseURL ?? process.env.DEEPSEEK_BASE_URL ?? PUBLIC_BASE_URL,
|
||||
defaults: {
|
||||
@@ -215,12 +246,12 @@ export function apply(ctx: Context, config: Config): void {
|
||||
const credentials = ctx.get('credentials')
|
||||
if (credentials !== undefined) {
|
||||
const hit = await credentials.resolve(ref)
|
||||
if (hit !== undefined) return hit.value
|
||||
if (hit !== undefined) return assertUsableApiKey(hit.value, 'llm-deepseek', ref)
|
||||
} else {
|
||||
// Without the seam, keep the historical ambient fallback so a plain
|
||||
// cordis.yml composition works from the environment alone.
|
||||
const ambient = process.env[ref]
|
||||
if (ambient !== undefined && ambient.length > 0) return ambient
|
||||
if (ambient !== undefined && ambient.length > 0) return assertUsableApiKey(ambient, 'llm-deepseek', ref)
|
||||
}
|
||||
throw new LlmError(
|
||||
`llm-deepseek: no API key for provider route "${PROVIDER}"; store ${ref} through the credentials`
|
||||
|
||||
@@ -2,8 +2,6 @@ import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService, { createUserMessage,
|
||||
CONTEXT_WINDOW_EXCEEDED_CODE,
|
||||
errorChain,
|
||||
LlmError,
|
||||
ProviderRequestId,
|
||||
QUOTA_EXCEEDED_CODE,
|
||||
ReasoningEffortId,
|
||||
@@ -206,18 +204,22 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('rejects a per-request effort before I/O when thinking is disabled', async () => {
|
||||
it('reports a per-request effort failure before I/O when thinking is disabled', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness(server.url, { thinking: 'disabled' })
|
||||
|
||||
await expect(assemble(ctx, {
|
||||
const result = await assemble(ctx, {
|
||||
model: 'deepseek-v4-flash',
|
||||
reasoningEffort: ReasoningEffortId('high'),
|
||||
messages: [createUserMessage({
|
||||
content: [{ type: 'text', text: 'hi' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
})],
|
||||
})).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' })
|
||||
})
|
||||
expect(result.finish).toMatchObject({
|
||||
kind: 'error',
|
||||
failure: { code: 'UNSUPPORTED_REASONING_EFFORT' },
|
||||
})
|
||||
expect(server.requests).toHaveLength(0)
|
||||
})
|
||||
|
||||
@@ -250,23 +252,22 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
[400, 'INVALID_REQUEST'],
|
||||
[500, 'SERVER'],
|
||||
[503, 'SERVER'],
|
||||
])('maps HTTP %d to LlmError code %s with the body message', async (status, code) => {
|
||||
])('maps HTTP %d to failure code %s with the body message', async (status, code) => {
|
||||
const behavior: Behavior = {
|
||||
kind: 'http-error',
|
||||
status,
|
||||
body: JSON.stringify({ error: { message: `failed with ${status}`, type: 't', code: 'c' } }),
|
||||
}
|
||||
const server = await mockServer([behavior, behavior])
|
||||
const server = await mockServer([behavior])
|
||||
const ctx = await harness(server.url)
|
||||
await expect(assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toThrow(`failed with ${status}`)
|
||||
await expect(
|
||||
assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] })
|
||||
.catch((error: unknown) => (error as LlmError).code),
|
||||
).resolves.toBe(code)
|
||||
const result = await assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toEqual({
|
||||
kind: 'error',
|
||||
failure: { message: `failed with ${status}`, code, status },
|
||||
})
|
||||
})
|
||||
|
||||
it('classifies a thrown HTTP context-window rejection with the canonical code', async () => {
|
||||
it('classifies an HTTP context-window failure with the canonical code', async () => {
|
||||
const server = await mockServer([{
|
||||
kind: 'http-error',
|
||||
status: 400,
|
||||
@@ -279,9 +280,11 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
}),
|
||||
}])
|
||||
const ctx = await harness(server.url)
|
||||
const code = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
.catch((error: unknown) => (error as LlmError).code)
|
||||
expect(code).toBe(CONTEXT_WINDOW_EXCEEDED_CODE)
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toMatchObject({
|
||||
kind: 'error',
|
||||
failure: { code: CONTEXT_WINDOW_EXCEEDED_CODE },
|
||||
})
|
||||
})
|
||||
|
||||
it('retains status, Retry-After seconds, and provider request id as structured facts', async () => {
|
||||
@@ -292,19 +295,16 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
headers: { 'retry-after': '2', 'x-request-id': 'req-429' },
|
||||
}])
|
||||
const ctx = await harness(server.url)
|
||||
let thrown: unknown
|
||||
try {
|
||||
await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
} catch (error: unknown) {
|
||||
thrown = error
|
||||
}
|
||||
expect(thrown).toBeInstanceOf(LlmError)
|
||||
expect((thrown as LlmError).failure).toEqual({
|
||||
message: 'slow down',
|
||||
code: 'RATE_LIMIT',
|
||||
status: 429,
|
||||
providerRetryAfterMs: 2_000,
|
||||
requestId: ProviderRequestId('req-429'),
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toEqual({
|
||||
kind: 'error',
|
||||
failure: {
|
||||
message: 'slow down',
|
||||
code: 'RATE_LIMIT',
|
||||
status: 429,
|
||||
providerRetryAfterMs: 2_000,
|
||||
requestId: ProviderRequestId('req-429'),
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
@@ -322,16 +322,17 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
},
|
||||
}])
|
||||
const ctx = await harness(server.url)
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toMatchObject({
|
||||
failure: {
|
||||
message: 'come back later',
|
||||
code: 'SERVER',
|
||||
status: 503,
|
||||
providerRetryAfterMs: 3_000,
|
||||
requestId: ProviderRequestId('deepseek-503'),
|
||||
},
|
||||
})
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toEqual({
|
||||
kind: 'error',
|
||||
failure: {
|
||||
message: 'come back later',
|
||||
code: 'SERVER',
|
||||
status: 503,
|
||||
providerRetryAfterMs: 3_000,
|
||||
requestId: ProviderRequestId('deepseek-503'),
|
||||
},
|
||||
})
|
||||
} finally {
|
||||
dateNow.mockRestore()
|
||||
}
|
||||
@@ -352,13 +353,11 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
headers: { 'retry-after': value },
|
||||
}])
|
||||
const ctx = await harness(server.url)
|
||||
let thrown: LlmError | undefined
|
||||
try {
|
||||
await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
} catch (error: unknown) {
|
||||
if (error instanceof LlmError) thrown = error
|
||||
}
|
||||
expect(thrown?.failure).toEqual({ message: 'retry later', code: 'RATE_LIMIT', status: 429 })
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toEqual({
|
||||
kind: 'error',
|
||||
failure: { message: 'retry later', code: 'RATE_LIMIT', status: 429 },
|
||||
})
|
||||
}
|
||||
})
|
||||
|
||||
@@ -379,53 +378,50 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
it('keeps the status-line message for JSON error bodies without a message', async () => {
|
||||
const server = await mockServer([{ kind: 'http-error', status: 500, body: '{"error":{"type":"x"}}' }])
|
||||
const ctx = await harness(server.url)
|
||||
await expect(assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toThrow(/HTTP 500/)
|
||||
const result = await assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish.kind).toBe('error')
|
||||
if (result.finish.kind !== 'error') throw new Error('expected an error finish')
|
||||
expect(result.finish.failure.code).toBe('SERVER')
|
||||
expect(result.finish.failure.message).toMatch(/HTTP 500/)
|
||||
})
|
||||
|
||||
it('keeps the status-line message for non-JSON error bodies', async () => {
|
||||
const server = await mockServer([{ kind: 'http-error', status: 502, body: 'Bad Gateway', contentType: 'text/plain' }])
|
||||
const ctx = await harness(server.url)
|
||||
await expect(assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toThrow(/HTTP 502/)
|
||||
const result = await assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish.kind).toBe('error')
|
||||
if (result.finish.kind !== 'error') throw new Error('expected an error finish')
|
||||
expect(result.finish.failure.code).toBe('SERVER')
|
||||
expect(result.finish.failure.message).toMatch(/HTTP 502/)
|
||||
})
|
||||
|
||||
it('maps unusual statuses to HTTP_<status>', () => {
|
||||
expect(httpErrorCode(418)).toBe('HTTP_418')
|
||||
})
|
||||
|
||||
it('wraps a transport failure in TRANSPORT with the fetch cause chain in the message', async () => {
|
||||
// Port 1 is reserved/unbound: fetch rejects with `TypeError: fetch failed`
|
||||
// whose actionable detail (ECONNREFUSED) lives on `cause`.
|
||||
it('reports a transport failure with the endpoint in the message', async () => {
|
||||
// Port 1 is reserved/unbound, so the service normalizes the fetch failure.
|
||||
const ctx = await harness('http://127.0.0.1:1')
|
||||
let caught: unknown
|
||||
try {
|
||||
await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
expect(caught).toBeInstanceOf(LlmError)
|
||||
const llmError = caught as LlmError
|
||||
expect(llmError.code).toBe('TRANSPORT')
|
||||
expect(llmError.message).toContain('http://127.0.0.1:1')
|
||||
expect(llmError.cause).toBeInstanceOf(TypeError)
|
||||
// The chain renderer reaches the transport diagnosis through the cause.
|
||||
expect(errorChain(llmError)).toMatch(/ECONNREFUSED|EADDRNOTAVAIL|bad port/)
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toMatchObject({
|
||||
kind: 'error',
|
||||
failure: {
|
||||
code: 'TRANSPORT',
|
||||
message: 'DeepSeek API request to http://127.0.0.1:1 failed',
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
it('classifies an aborted request without losing the transport rejection', async () => {
|
||||
it('classifies an aborted request as an aborted finish', async () => {
|
||||
const controller = new AbortController()
|
||||
controller.abort()
|
||||
const ctx = await harness('http://127.0.0.1:1')
|
||||
let caught: unknown
|
||||
try {
|
||||
await assemble(ctx, { model: 'deepseek-v4-flash', messages: [], signal: controller.signal })
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
expect(caught).toBeInstanceOf(LlmError)
|
||||
expect(caught).toMatchObject({ code: 'ABORTED' })
|
||||
expect((caught as LlmError).cause).toMatchObject({ name: 'AbortError' })
|
||||
const result = await assemble(ctx, {
|
||||
model: 'deepseek-v4-flash',
|
||||
messages: [],
|
||||
signal: controller.signal,
|
||||
})
|
||||
expect(result.finish).toMatchObject({ kind: 'aborted', failure: { code: 'ABORTED' } })
|
||||
})
|
||||
|
||||
it('throws EMPTY_RESPONSE when the response has no body', async () => {
|
||||
@@ -443,20 +439,17 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
}
|
||||
})
|
||||
|
||||
it('classifies an abrupt body close as TRANSPORT and retains its cause', async () => {
|
||||
it('classifies an abrupt body close as TRANSPORT', async () => {
|
||||
const server = await mockServer([{
|
||||
kind: 'close-early',
|
||||
events: ['{"choices":[{"delta":{"content":"par"}}]}'],
|
||||
}])
|
||||
const ctx = await harness(server.url)
|
||||
let caught: unknown
|
||||
try {
|
||||
await assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] })
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
expect(caught).toMatchObject({ code: 'TRANSPORT' })
|
||||
expect(errorChain(caught)).toMatch(/terminated|socket|without \[DONE\]/)
|
||||
const result = await assemble(ctx,{ model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish.kind).toBe('error')
|
||||
if (result.finish.kind !== 'error') throw new Error('expected an error finish')
|
||||
expect(result.finish.failure.code).toBe('TRANSPORT')
|
||||
expect(result.finish.failure.message).toMatch(/^DeepSeek API stream from .* failed$/)
|
||||
})
|
||||
|
||||
it('aborts mid-stream via the request signal', async () => {
|
||||
@@ -478,7 +471,13 @@ describe('DeepSeekAdapter against a mock server', () => {
|
||||
})()
|
||||
|
||||
setTimeout(() => { controller.abort() }, 30)
|
||||
await expect(pending).rejects.toMatchObject({ code: 'ABORTED' })
|
||||
const chunks = await pending
|
||||
expect(chunks).toHaveLength(1)
|
||||
expect(chunks[0]?.type).toBe('finish')
|
||||
if (chunks[0]?.type !== 'finish') throw new Error('expected a finish chunk')
|
||||
expect(chunks[0].reason.kind).toBe('aborted')
|
||||
if (chunks[0].reason.kind !== 'aborted') throw new Error('expected an aborted finish')
|
||||
expect(chunks[0].reason.failure.code).toBe('ABORTED')
|
||||
})
|
||||
|
||||
it('maps connection failures to TRANSPORT without losing the cause', async () => {
|
||||
@@ -700,6 +699,13 @@ describe('plugin registration and config', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('normalizes a literal API key and treats whitespace as absent', () => {
|
||||
expect(resolveAdapterOptions({ apiKey: ' key ' }).apiKey).toBe('key')
|
||||
const whitespace = resolveAdapterOptions({ apiKey: ' \t ', apiKeyEnv: 'CUSTOM_API_KEY' })
|
||||
expect(whitespace.apiKey).toBeUndefined()
|
||||
expect(whitespace.apiKeyEnv).toBe('CUSTOM_API_KEY')
|
||||
})
|
||||
|
||||
it('uses the default model catalog when apply is called directly', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
@@ -793,6 +799,26 @@ describe('plugin registration and config', () => {
|
||||
expect(ctx.llm.listProviders()).toEqual([])
|
||||
})
|
||||
|
||||
it.each([0, 1.5])('rejects a per-model output cap of %s', (maxTokens) => {
|
||||
expect(() => resolveAdapterOptions({ models: [{ id: 'bad-cap', maxTokens }] }))
|
||||
.toThrow(/maxTokens must be a positive integer/)
|
||||
})
|
||||
|
||||
it('prefers a model\'s own output cap over the profile default', async () => {
|
||||
// The profile default stays what an unlisted or uncapped model resolves
|
||||
// to, so adding a per-model cap changes one model rather than the route.
|
||||
const adapter = adapterOf({ maxTokens: 4096, models: [
|
||||
{ id: 'capped', maxTokens: 512 },
|
||||
{ id: 'uncapped' },
|
||||
] })
|
||||
await expect(adapter.resolveModel('deepseek-official', 'capped'))
|
||||
.resolves.toMatchObject({ defaultMaxTokens: 512 })
|
||||
await expect(adapter.resolveModel('deepseek-official', 'uncapped'))
|
||||
.resolves.toMatchObject({ defaultMaxTokens: 4096 })
|
||||
await expect(adapter.resolveModel('deepseek-official', 'not-in-catalog'))
|
||||
.resolves.toMatchObject({ defaultMaxTokens: 4096 })
|
||||
})
|
||||
|
||||
it('rejects invalid context capacity when apply is called directly', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
@@ -858,12 +884,15 @@ describe('plugin registration and config', () => {
|
||||
// only the request itself needs a key.
|
||||
expect(ctx.llm.listProviders()).toEqual([{ id: 'deepseek-official', name: 'DeepSeek' }])
|
||||
await expect(ctx.llm.listModels('deepseek-official')).resolves.toHaveLength(2)
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toMatchObject({ code: 'MISSING_CREDENTIAL' })
|
||||
const first = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(first.finish).toMatchObject({ kind: 'error', failure: { code: 'MISSING_CREDENTIAL' } })
|
||||
// The guidance leads with the credential store — the path that keeps the
|
||||
// secret out of configuration files — and mentions a literal key last.
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toThrow(/store DEEPSEEK_API_KEY through the credentials service.*as a last resort.*"apiKey"/s)
|
||||
const second = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(second.finish.kind).toBe('error')
|
||||
if (second.finish.kind !== 'error') throw new Error('expected an error finish')
|
||||
expect(second.finish.failure.message)
|
||||
.toMatch(/store DEEPSEEK_API_KEY through the credentials service.*as a last resort.*"apiKey"/s)
|
||||
})
|
||||
|
||||
it('reads the ambient variable when no credentials seam is mounted', async () => {
|
||||
@@ -883,8 +912,8 @@ describe('plugin registration and config', () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(LlmDeepSeek, { baseURL: 'http://127.0.0.1:1' })
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toMatchObject({ code: 'MISSING_CREDENTIAL' })
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toMatchObject({ kind: 'error', failure: { code: 'MISSING_CREDENTIAL' } })
|
||||
})
|
||||
|
||||
it('prefers explicit config over env for key and base URL', async () => {
|
||||
@@ -969,3 +998,38 @@ describe('plugin registration and config', () => {
|
||||
expect(ctx.llm.listProviders()).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
describe('API key format', () => {
|
||||
it('trims a padded literal apiKey', () => {
|
||||
expect(resolveAdapterOptions({ apiKey: ' sk-abc ' }).apiKey).toBe('sk-abc')
|
||||
})
|
||||
|
||||
it('leaves an omitted apiKey absent so apiKeyEnv still resolves it', () => {
|
||||
expect(resolveAdapterOptions({}).apiKey).toBeUndefined()
|
||||
})
|
||||
|
||||
it('treats a whitespace-only literal apiKey as absent, not as a failure', () => {
|
||||
// This adapter's absence has a defined fallback, so a blank literal
|
||||
// resolves through apiKeyEnv like an omitted one. (llm-pi-ai refuses a
|
||||
// blank one instead: there, absence selects provider-native or OAuth
|
||||
// authentication rather than a different source for the same key.)
|
||||
const resolved = resolveAdapterOptions({ apiKey: ' ', apiKeyEnv: 'CUSTOM_API_KEY' })
|
||||
expect(resolved.apiKey).toBeUndefined()
|
||||
expect(resolved.apiKeyEnv).toBe('CUSTOM_API_KEY')
|
||||
})
|
||||
|
||||
it('rejects a literal apiKey no header can carry', () => {
|
||||
expect(() => resolveAdapterOptions({ apiKey: 'sk-\u{1F600}' }))
|
||||
.toThrow(/no HTTP header can carry/)
|
||||
})
|
||||
|
||||
it('never echoes the key in the rejection', () => {
|
||||
const secret = 'sk-\u{1F600}supersecret'
|
||||
expect(() => resolveAdapterOptions({ apiKey: secret })).toThrow()
|
||||
try {
|
||||
resolveAdapterOptions({ apiKey: secret })
|
||||
} catch (error) {
|
||||
expect((error as Error).message).not.toContain('supersecret')
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
@@ -3,7 +3,7 @@ import { Context } from 'cordis'
|
||||
import { mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import LlmService from '@deepseek-ai/dsh-llm'
|
||||
import LlmService, { INVALID_CREDENTIAL_CODE } from '@deepseek-ai/dsh-llm'
|
||||
import { credentialRef } from '@deepseek-ai/dsh-credentials'
|
||||
import { CredentialsLocal } from '@deepseek-ai/dsh-credentials-local'
|
||||
import { settingsNamespace } from '@deepseek-ai/dsh-settings'
|
||||
@@ -96,12 +96,32 @@ describe('request-level dynamic configuration', () => {
|
||||
const server = await mockServer([{ kind: 'sse', events: textEvents }])
|
||||
const { ctx } = await boot(dir, { baseURL: server.url })
|
||||
|
||||
await expect(prompt(ctx)).rejects.toMatchObject({ code: 'MISSING_CREDENTIAL' })
|
||||
const keyless = await prompt(ctx)
|
||||
expect(keyless.finish).toMatchObject({ kind: 'error', failure: { code: 'MISSING_CREDENTIAL' } })
|
||||
await ctx.credentials.set(KEY_REF, 'sk-arrived')
|
||||
await prompt(ctx)
|
||||
expect(server.headers[0]?.authorization).toBe('Bearer sk-arrived')
|
||||
})
|
||||
|
||||
it('rejects a stored credential no header can carry, never echoing it in the failure', async () => {
|
||||
vi.stubEnv('DEEPSEEK_API_KEY', '')
|
||||
const dir = await home()
|
||||
const { ctx } = await boot(dir, { baseURL: 'http://127.0.0.1:1' })
|
||||
const secret = 'sk-\u{1F600}supersecret'
|
||||
|
||||
// The real credentials seam (the path the web Models page writes through),
|
||||
// not a hand-built stub: this package's own dynamic-config harness already
|
||||
// boots one, and round-tripping the value through its actual store/read
|
||||
// path is stronger evidence than a canned in-memory return would be.
|
||||
await ctx.credentials.set(KEY_REF, secret)
|
||||
const result = await prompt(ctx)
|
||||
expect(result.finish).toMatchObject({ kind: 'error', failure: { code: INVALID_CREDENTIAL_CODE } })
|
||||
if (result.finish.kind !== 'error') throw new Error('expected an error finish')
|
||||
expect(result.finish.failure.message).not.toContain(secret)
|
||||
expect(result.finish.failure.message).not.toContain('supersecret')
|
||||
expect(result.finish.failure.message).not.toContain('ByteString')
|
||||
})
|
||||
|
||||
it('advertises a live settings catalog without re-registration', async () => {
|
||||
const dir = await home()
|
||||
const { ctx } = await boot(dir, { apiKey: 'k', baseURL: 'http://127.0.0.1:1' })
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/llm-pi-ai/README.md
|
||||
README.md: 7f6483568aec75fdfa457d21d370b289116a4bf4
|
||||
README.zh.md: accd69ff5e687d76a1ff552972bf60455307d567
|
||||
README.md: 97bd629adedda9d63fee730bc31129b0c22cc704
|
||||
README.zh.md: 71d45b590f48f4b8162ae329b58b5ff4a9eb13b1
|
||||
|
||||
@@ -2,19 +2,20 @@
|
||||
|
||||
English | [中文](README.zh.md)
|
||||
|
||||
Generic multi-provider adapter for the harness LLM seam backed by [`@earendil-works/pi-ai`](https://www.npmjs.com/package/@earendil-works/pi-ai). One plugin instance owns a dict of provider profiles keyed by route; every request selects a profile with `GenerateOptions.provider` and resolves `GenerateOptions.model` dynamically from pi-ai's installed catalog.
|
||||
Generic multi-provider adapter for the harness LLM seam backed by [`@earendil-works/pi-ai`](https://www.npmjs.com/package/@earendil-works/pi-ai). One plugin instance owns a dict of provider profiles keyed by route; every request selects a profile with `GenerateOptions.provider` and resolves `GenerateOptions.model` against that route's configured catalog. A route naming an installed pi-ai provider inherits its endpoint, wire protocol, and model catalog as defaults and overrides them field by field; a route pi-ai does not ship is declared outright, so an OpenAI-compatible gateway, a self-hosted server, or a provider newer than the installed catalog is configuration rather than a code change.
|
||||
|
||||
The package root exposes the Cordis plugin contract and `PiAiAdapter`; profile resolution, model construction, replay conversion, and stream conversion remain package-internal.
|
||||
The package root exposes the Cordis plugin contract, `PiAiAdapter`, and `supportedProtocols()`; profile resolution, catalog materialization, provider construction, replay conversion, and stream conversion remain package-internal.
|
||||
|
||||
## Config
|
||||
|
||||
Configure credentials and deployment-specific transport settings per provider, keyed by the provider route itself. Prefer `apiKeyEnv` — a credential *reference* resolved per request — over a literal `apiKey`, so no secret enters this file. Omitting **both** is what delegates authentication to pi-ai's provider-native ambient discovery; a configured reference that resolves to nothing fails the request with `MISSING_CREDENTIAL` instead, because falling through would authenticate with whatever unrelated key the environment happens to hold. `baseURL` overrides only the endpoint of the selected catalog model, preserving its API family and compatibility metadata, so private proxies such as `https://proxy.example.com:8443` remain supported.
|
||||
Configure credentials, the model catalog, and deployment-specific transport settings per provider, keyed by the provider route itself. Prefer `apiKeyEnv` — a credential *reference* resolved per request — over a literal `apiKey`, so no secret enters this file. Omitting **both** is what leaves the route unauthenticated, which for an installed catalog route means pi-ai's provider-native ambient discovery; a configured reference that resolves to nothing fails the request with `MISSING_CREDENTIAL` instead, because falling through would authenticate with whatever unrelated key the environment happens to hold. One credential serves every model on its route.
|
||||
|
||||
```yaml
|
||||
- id: llm
|
||||
name: '@deepseek-ai/dsh-llm-pi-ai'
|
||||
config:
|
||||
providers:
|
||||
# Catalog route: endpoint, protocol, and models all come from pi-ai.
|
||||
openai:
|
||||
apiKeyEnv: OPENAI_API_KEY
|
||||
baseURL: https://proxy.example.com:8443
|
||||
@@ -26,36 +27,77 @@ Configure credentials and deployment-specific transport settings per provider, k
|
||||
initialDelayMs: 500
|
||||
maxDelayMs: 10000
|
||||
jitterRatio: 0.1
|
||||
# Catalog route with its catalog narrowed to one model and that model's
|
||||
# capacity corrected; every unset field still comes from the catalog.
|
||||
anthropic:
|
||||
apiKeyEnv: ANTHROPIC_API_KEY
|
||||
streamIdleTimeoutMs: 300000
|
||||
openrouter:
|
||||
apiKeyEnv: OPENROUTER_API_KEY
|
||||
headers:
|
||||
X-Deployment: production
|
||||
models:
|
||||
- id: claude-sonnet-4-5
|
||||
contextWindow: 200000
|
||||
# Hand-declared route: pi-ai ships nothing under this key, so the profile
|
||||
# supplies the whole provider.
|
||||
acme-gateway:
|
||||
displayName: Acme Gateway
|
||||
apiKeyEnv: ACME_GATEWAY_API_KEY
|
||||
api: openai-completions
|
||||
baseURL: https://gateway.acme.example/v1
|
||||
models:
|
||||
- id: acme-large
|
||||
name: Acme Large
|
||||
contextWindow: 65536
|
||||
maxTokens: 4096
|
||||
```
|
||||
|
||||
Each dict key must exist in pi-ai's installed catalog; the dict shape makes duplicates unrepresentable, and the pre-release array shape (with per-profile `provider` fields) fails load with migration directions. `providers` may also be empty or omitted entirely: the adapter then mounts **dormant** — zero routes, no extra catalog entries — and registers routes the moment the `llm-pi-ai:` settings section supplies profiles, dropping them again when it empties. Dormant or not, the plugin declares every installed catalog provider in the configurable-provider directory (`ctx.llm.listConfigurableProviders()`, settings path `providers.<provider>`), so configuration surfaces can offer the full catalog before any route exists. Which adapters exist is composition; which providers run can be entirely the user's settings document. Registration with `ctx.llm` is atomic: a collision with any provider route already owned by another adapter fails plugin loading without registering the remaining routes. Model ids are not lifecycle config; an unknown model fails before any provider request with `LlmError('UNKNOWN_MODEL')`.
|
||||
The dict shape makes duplicate routes unrepresentable, and the pre-release array shape (with per-profile `provider` fields) fails load with migration directions. `providers` may also be empty or omitted entirely: the adapter then mounts **dormant** — zero routes, no extra catalog entries — and registers routes the moment the `llm-pi-ai:` settings section supplies profiles, dropping them again when it empties. Dormant or not, the plugin declares every installed catalog provider in the configurable-provider directory (`ctx.llm.listConfigurableProviders()`, settings path `providers.<provider>`), joined with every route the current profiles declare, so configuration surfaces can offer the full catalog before any route exists and can still address a hand-declared one. Each entry carries `declared`: whether pi-ai ships nothing under that key. It follows the installed catalog, never the settings document, because narrowing a shipped provider's models stores a profile too and that route is still one pi-ai knows — only the adapter can tell the two apart, which is why the directory answers rather than leaving a surface to infer it. Which adapters exist is composition; which providers run can be entirely the user's settings document. Registration with `ctx.llm` is atomic: a collision with any provider route already owned by another adapter fails plugin loading without registering the remaining routes. Model ids are not lifecycle config; a model the route does not configure fails before any provider request with `LlmError('UNKNOWN_MODEL')`.
|
||||
|
||||
## Catalog resolution
|
||||
|
||||
A profile's `models` list *replaces* the route's installed catalog rather than extending it; omitting it (or leaving it empty) serves that catalog unchanged. Each entry defaults its unset fields from the installed model of the same `id`, so narrowing a catalog route to two models, correcting one capacity, or adding a model newer than the installed catalog are all one-line edits. Only the fields the harness consumes are configurable — `id`, `name`, `contextWindow`, and `maxTokens`. Pricing and input modalities have no harness consumer and ride the installed entry or are absent. Reasoning is not per-model configurable at all: a bare capability flag would make pi-ai advertise effort levels with no `thinkingLevelMap` to spell them, and no listing endpoint reports a model's reasoning protocol, so reasoning rides the installed catalog entry or is absent.
|
||||
|
||||
A model neither the entry nor the installed catalog sizes takes the route's `defaultContextWindow` (262,144) and `defaultMaxTokens` (32,768), so a listing that discloses nothing but ids still yields a serviceable route. Both fallbacks are guesses by construction, which is why they are route fields a deployment whose gateway serves smaller models corrects once rather than constants buried in the adapter; the fallback sizes the model and never becomes a per-request cap.
|
||||
|
||||
Resolution still fails loud, naming the offending route and model, when a route cannot be served at all: a route the catalog does not ship needs `api`, `baseURL`, and a non-empty `models` list of uniquely-identified models. That resolution runs inside the section schema, so an unserviceable profile is refused **where it is written** — `settings.mutate` answers `settings-rejected` naming the route and model — rather than being stored and then quietly disabling every route in the namespace. The settings seam keeps a namespace's last good value for an already-stored section that fails, so this cannot strand a deployment. `api` accepts the protocols in `supportedProtocols()` and is only needed when the catalog cannot supply one: a model absent from the catalog inherits the protocol its shipped siblings agree on, so adding a model to a single-protocol catalog route restates nothing.
|
||||
|
||||
`baseURL` sets the endpoint of every model on the route, so private proxies such as `https://proxy.example.com:8443` remain supported; a catalog route that omits it keeps each catalog model's own endpoint. Naming `api` on a catalog route repoints the whole route at that protocol, which is how a deployment moves a provider between, say, Responses and Chat Completions.
|
||||
|
||||
`supportedProtocols()` is deliberately narrower than pi-ai's full streaming API set: it holds only the protocols a profile can *completely* describe with a key, an endpoint, and headers. Bedrock signs with SigV4 over AWS credentials and a region, Vertex needs a project, a location, and application-default credentials, Azure needs provider environment plus an api-version, and Codex authenticates through OAuth — offering those would hand back a route that cannot authenticate. Catalog routes still reach them through their own provider; only an explicit override is refused.
|
||||
|
||||
## Dynamic configuration (settings + credentials)
|
||||
|
||||
The adapter reads its profiles through a thunk **once per operation** instead of freezing them at construction. The plugin registers the `llm-pi-ai` namespace on the optional `ctx.settings` seam with this same `Config` schema and its `cordis.yml` entry as the composition `base`, and because `providers` is a dict, the base and the user's `llm-pi-ai:` settings section merge **per provider**: a user can add a route, override one field of a composition route, or point a route at another proxy, all effective on the next request with no restart. Without a mounted settings service the entry config alone drives the adapter, unchanged.
|
||||
|
||||
Credentials resolve per stream call: a non-empty literal `apiKey` wins, then `apiKeyEnv` through the optional `ctx.credentials` seam (`$DSH_HOME/.env` under the live environment; exactly that variable without a mounted seam). A profile naming no credential at all — and only that case — defers to pi-ai's ambient discovery. The route set and each route's captured retry policy are the registration-level facts: when either changes, the plugin replaces its registration atomically (same adapter instance, candidate set validated first), so a route another adapter already owns leaves the previous routes serving and reverting to a working configuration re-applies. Provider key order never counts as a change. A live settings snapshot naming an unknown provider (or failing any other resolver bound) keeps the last good profiles and logs the failure; the entry config itself still fails plugin load.
|
||||
Credentials resolve per stream call: a non-empty literal `apiKey` wins, then `apiKeyEnv` through the optional `ctx.credentials` seam (`$DSH_HOME/.env` under the live environment; exactly that variable without a mounted seam). A profile naming no credential at all — and only that case — defers to pi-ai's ambient discovery. Every key is trimmed and format-checked before use — a literal `apiKey` when profiles resolve (plugin load, or the next settings snapshot), a value `apiKeyEnv` resolves at request time — so a value no HTTP header can carry is refused there instead of surfacing as an opaque `fetch` `TypeError`; the request-time refusal throws `LlmError('INVALID_CREDENTIAL')` naming the failing route and credential reference but never any part of the key. The route set and each route's captured retry policy are the registration-level facts: when either changes, the plugin replaces its registration atomically (same adapter instance, candidate set validated first), so a route another adapter already owns leaves the previous routes serving and reverting to a working configuration re-applies. Provider key order never counts as a change. A section this adapter could not serve is refused where it is written — the registered `validate` resolves the whole profile set, so `ctx.settings.mutate` rejects with the resolver's own error (the wire surface reports it as `settings-rejected`) and nothing is stored. A stored section that becomes unserviceable some other way — an external edit of `settings.yaml` — keeps the namespace's last good value at the settings seam and warns. The entry config itself still fails plugin load, and a route the llm registry refuses (one another adapter family already owns) is logged while the previously registered routes keep serving.
|
||||
|
||||
The adapter exposes each configured provider's installed pi-ai models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata derived from `getModels(provider)`; request-time resolution still performs the authoritative catalog lookup, so discovery does not create a second model registry. `ctx.llm.resolveModelInfo(provider, model)` performs that exact descriptor lookup once and returns its identity, context window, and selectable thinking levels, keeping authoritative metadata on the route-owning adapter rather than its consumers.
|
||||
The adapter exposes each configured route's models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata read from the same pi-ai `Models` collection the request path uses, so discovery does not create a second model registry. `ctx.llm.resolveModelInfo(provider, model)` performs that exact descriptor lookup once and returns its identity, context window, configured output cap, and selectable thinking levels, keeping authoritative metadata on the route-owning adapter rather than its consumers. A model's **configured** `maxTokens` becomes the seam's `defaultMaxTokens`, so a request that names no output cap carries the one the deployment chose; a value inherited from the installed catalog is the model's output *capability* and never becomes a request default on its own.
|
||||
|
||||
The `reasoning.efforts` list is pi-ai's ordered `getSupportedThinkingLevels(model)` result without filtering or normalization, including `off` and the model-specific availability of `xhigh` or `max`. The Harness exposes each canonical pi-ai level as an opaque ID; provider/model wire spellings remain inside pi-ai's `thinkingLevelMap`. A non-reasoning model therefore exposes pi-ai's `off` choice. The profile `reasoning` value, including `off`, is the deployment default when configured; omitting it preserves the provider default. Per-request `GenerateOptions.reasoningEffort` takes precedence, and any explicit value absent from the exact model capability fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O instead of being clamped. pi-ai's common stream options represent `off` by omitting `reasoning`.
|
||||
A model that carries reasoning metadata exposes pi-ai's ordered `getSupportedThinkingLevels(model)` result without filtering or normalization, including `off` and the model-specific availability of `xhigh` or `max`. The Harness exposes each canonical pi-ai level as an opaque ID; provider/model wire spellings remain inside pi-ai's `thinkingLevelMap`.
|
||||
|
||||
Supported profile fields are `apiKey`, `apiKeyEnv`, `baseURL`, `headers`, `reasoning`, `thinkingBudgets`, `cacheRetention`, `transport`, `timeoutMs`, `websocketConnectTimeoutMs`, `streamIdleTimeoutMs`, and `retryPolicy`. Each profile's optional retry policy is captured with that provider route; omission uses bounded normal defaults. The stream-idle interval is a positive finite Node timer delay, defaults to five minutes, and covers only an outstanding provider read, not consumer think time. Harness app attribution wins a conflicting configured header name.
|
||||
A model **without** that metadata — every hand-declared one, and a catalog model pi-ai marks as non-reasoning — exposes no `reasoning` at all. pi-ai reports such a model as supporting the single level `off`, but `off` is translated to *omitting* the reasoning option, which is byte-for-byte the request that naming no effort already produces: selecting it could not disable anything, so a provider whose own default is to think would keep thinking with `off` shown as selected. Reporting the capability as unavailable leaves a surface offering the provider's default and nothing that misrepresents it. The profile `reasoning` value, including `off`, is the deployment default when configured; omitting it preserves the provider default. Per-request `GenerateOptions.reasoningEffort` takes precedence, and a level absent from the exact model capability fails the REQUEST with `UNSUPPORTED_REASONING_EFFORT` before network I/O instead of being clamped. Describing a model never fails that way: the models under one provider disagree about which levels they accept, so `resolveModel` reports a profile level the exact model cannot take as no default at all rather than throwing. A throw there would take the whole provider out of every model catalog built over it — one mis-set profile field hiding even the models that do support the level — so a bad configuration surfaces where it is acted on, not where it is described. pi-ai's common stream options represent `off` by omitting `reasoning`.
|
||||
|
||||
Supported profile fields are `apiKey`, `apiKeyEnv`, `displayName`, `api`, `baseURL`, `models`, `defaultContextWindow`, `defaultMaxTokens`, `headers`, `reasoning`, `thinkingBudgets`, `cacheRetention`, `transport`, `timeoutMs`, `websocketConnectTimeoutMs`, `streamIdleTimeoutMs`, and `retryPolicy`. Each profile's optional retry policy is captured with that provider route; omission uses bounded normal defaults. The stream-idle interval is a positive finite Node timer delay, defaults to five minutes, and covers only an outstanding provider read, not consumer think time. Harness app attribution wins a conflicting configured header name.
|
||||
|
||||
The adapter forces pi-ai's SDK `maxRetries` to zero so one `stream()` call makes one provider request. The removed profile fields `maxRetries` and `maxRetryDelayMs` fail load instead of silently multiplying or hiding the separately composed agent-level retry budget. Idle expiry aborts the SDK's stable request signal and surfaces `TIMEOUT`; an earlier caller abort remains `ABORTED`.
|
||||
|
||||
Image requests resolve the optional `ctx.attachments` service when the request is dispatched, so Cordis plugin load order does not freeze attachment availability. Image detection and conversion recurse through nested `tool-result` content, so a nested image is neither flattened nor skipped. A visual request still fails explicitly with `UNSUPPORTED_CONTENT` when the service or the selected model's image capability is absent.
|
||||
## Endpoint interrogation
|
||||
|
||||
The plugin offers `ctx.llm.registerModelDiscovery('llm-pi-ai', …)`, which answers "which models can this provider serve?" for a route a configuration surface is editing or drafting. It is deliberately *not* a catalog refresh: nothing is stored, and the reply is candidates the surface offers for adoption. `settings.yaml` remains the only thing that decides what a route serves.
|
||||
|
||||
A request naming a route the **installed catalog ships is answered from that catalog**, with no network call: pi-ai's registry is the authoritative list for its own providers, and it carries the context windows and output caps a listing endpoint would not disclose. Such a route needs no `baseURL` at all. Only a route the catalog does not describe — a gateway, a self-hosted server — is interrogated over the wire, and one that names no endpoint is told to set one or enter its models by hand.
|
||||
|
||||
A draft carries the credential the user typed, if any; a route that already stored one shows a configuration surface only a redacted descriptor, so the interrogation supplies that route's own credential — resolved exactly as a request to it would, `apiKey` then `apiKeyEnv` — rather than going out unauthenticated and reporting the endpoint's 401 as a wrong key. A typed key wins, being the one under test. Resolution happens only on the path that reaches the network, so a catalog route answers without touching credentials at all. A supplied or stored probe key is trimmed and format-checked the same way, so a value no HTTP header can carry is refused immediately as `LlmError('INVALID_CREDENTIAL')` instead of reaching `fetch`, where it would surface as an opaque `ByteString` failure indistinguishable from an unreachable endpoint.
|
||||
|
||||
Interrogation reads `openai-completions` and `openai-responses`, whose `GET /models` shape with bearer auth is the one a gateway, a self-hosted server, and the official endpoints all agree on. Azure is excluded despite its OpenAI lineage — it authenticates with an `api-key` header and requires an `api-version` query — and Codex uses OAuth; every other protocol answers `DISCOVERY_UNSUPPORTED` so the surface falls back to hand-entry instead of an authentication failure being reported as a provider with no models. The `baseURL` is treated as a prefix rather than a URL to resolve against, so a deployment path such as `https://gateway.example/openai/v1` keeps its segments.
|
||||
|
||||
Most listings disclose an id and nothing else; `context_window`/`context_length` and `max_output_tokens`/`max_tokens` are read when a gateway supplies them, entries without a usable id are skipped rather than failing the whole listing, and everything else the adopting surface still owes. The reply is read under a four-megabyte ceiling enforced on the bytes actually received — the endpoint is a URL the user typed, so a declared length is checked first but never trusted as the bound. An unreachable endpoint, a refused credential, a non-JSON body, and a body with no `data` array all fail with `DISCOVERY_FAILED` and a message naming the endpoint and, for a 401 or 403 alone, the credential. Cancellation during the body read surfaces as `ABORTED`, like a cancellation before the request went out.
|
||||
|
||||
## Provider/model routing and replay
|
||||
|
||||
The selected pi-ai catalog descriptor supplies the protocol implementation. This includes native API differences such as OpenAI models whose descriptor uses the Responses API rather than Chat Completions; the harness adapter does not hardcode endpoint selection by model name.
|
||||
Each resolution produces one **immutable** snapshot — the profiles plus a `createModels()` collection holding the `Provider` each route built — and every operation captures a whole snapshot before its first `await`. A configuration change builds a *new* collection rather than mutating the one in use: `Models.streamSimple()` resolves its provider lazily, when the stream is first consumed, which is after the credential await, so a mutated collection would let a request that started under one configuration finish under another or fail on a provider that no longer exists. This is what makes the seam's per-step call freeze (`llm.prepareCall()`) hold end to end — switching models mid-reply takes effect on the next step, never inside the one in flight. Requests reach their provider through `Models.streamSimple()`. A catalog route that keeps its catalog protocol **reuses** the installed provider with its model list replaced, because that provider owns API implementations this package cannot reconstruct — Bedrock loads its Smithy module through a separate entry point — so rebuilding it from parts would silently narrow which providers work. Every other route is built by `createProvider()` over the protocol table behind `supportedProtocols()`, whose entries are the same factories pi-ai's own provider factories use.
|
||||
|
||||
Credentials never enter that collection. The harness resolves a route's key through its own seam before the request reaches pi-ai and passes it as the request's `apiKey` option, which pi-ai treats as the highest-priority auth override; `Models` therefore holds no credential store, and the harness keeps its fail-loud reference semantics. A route naming no credential resolves as configured-but-keyless and leaves the requirement to the protocol, which is where it actually lives.
|
||||
|
||||
The selected model descriptor supplies the protocol implementation. This includes native API differences such as OpenAI models whose descriptor uses the Responses API rather than Chat Completions; the harness adapter does not hardcode endpoint selection by model name.
|
||||
|
||||
Successful assistant responses store a versioned, lossless-JSON replay state beside their durable provider/model provenance. At request time, `LlmService` passes replay state only when the historical provider route and target provider route are currently owned by this same `PiAiAdapter` instance. The adapter validates the state and restores pi-ai response ids and provider signatures even when the target provider or model changes; pi-ai then decides which metadata its target API can reuse. History without replay state is translated as foreign provider-neutral content and never impersonates a native pi-ai response.
|
||||
|
||||
@@ -77,10 +119,6 @@ Every request carries the shared attribution header from dsh-llm's `attributionH
|
||||
|
||||
pi-ai installs several provider SDKs and lazy-loads the one selected by the catalog model. The dependency weight is isolated to this opt-in adapter package.
|
||||
|
||||
## Testing
|
||||
|
||||
Unit tests use pi-ai catalog models redirected to local mock servers and cover provider/profile routing, one wire request per adapter call, idle-timeout response termination, caller abort, native API selection, endpoint overrides, attribution, conversion, replay-state validation, and cross-provider/model replay within one adapter instance. `tests/dynamic-config.spec.ts` drives real settings-local and credentials-local providers: a settings-born route registers live and drops when the user layer resets, `apiKeyEnv` credentials rotate between requests, and an unknown-provider snapshot keeps the last good profiles. `tests/loader-composition.spec.ts` boots the dormant posture from a test-only `cordis.yml` through the actual Loader and registers its route from an on-disk `settings.yaml` edit. Real-API coverage remains key-gated under `pnpm run test:e2e`.
|
||||
|
||||
## Model Experience
|
||||
|
||||
### Provider request through pi-ai
|
||||
@@ -115,7 +153,9 @@ Recorded response content appends to the next request and does not invalidate it
|
||||
|
||||
- **Settings can add or override routes, not remove composition routes** — the user layer merges over the composition `base`, so deleting a `cordis.yml`-provided provider is a composition change; `replace` on the namespace only resets the user layer.
|
||||
- **`headers` can carry a credential the redactor never sees** — the profile's `headers` dict is plain strings, so `Authorization` or `api-key` set there is returned verbatim by a redacted `describe()` and rendered by any configuration UI. Store credentials as `apiKeyEnv` references; making the dict write-only is deferred with the rest of the [wire-boundary work](../llm/README.md#known-limitations-and-deferred-work).
|
||||
- **Catalog membership is required** — custom model ids that are absent from the installed pi-ai catalog fail with `UNKNOWN_MODEL`, even when a provider profile supplies a custom endpoint.
|
||||
- **A route's catalog never refreshes itself** — the catalog is whatever `settings.yaml` says, so a model list is only as current as its last edit. Nothing here queries a provider for the models it serves; a route gains a model when someone writes one.
|
||||
- **One wire protocol per route** — `api` applies to the whole route, so a mixed-protocol catalog route (an OpenAI-style catalog spanning Responses and Chat Completions) cannot host a model of the other protocol, and adding a model such a route does not describe requires naming `api` and moving every model onto it. Splitting the provider across two route keys is the workaround.
|
||||
- **An unauthenticated route depends on its protocol** — naming no credential resolves the route as configured-but-keyless, but pi-ai's OpenAI-compatible implementation still requires an API key or an `Authorization` header, so a keyless local server needs a placeholder `apiKey` or an `Authorization` entry in `headers`.
|
||||
- **`GenerateOptions.stop` is unsupported** — pi-ai's common stream options cannot guarantee stop-sequence behavior across providers, so the adapter rejects the field.
|
||||
- **In-history `system` messages use pi-ai's common context conversion** — provider-specific placement follows pi-ai rather than a harness-owned wire override.
|
||||
- **Provider HTTP status is unavailable** — pi-ai error events do not expose a stable HTTP status across providers; failures expose only stable harness error codes.
|
||||
|
||||
@@ -2,19 +2,20 @@
|
||||
|
||||
[English](README.md) | 中文
|
||||
|
||||
基于 [`@earendil-works/pi-ai`](https://www.npmjs.com/package/@earendil-works/pi-ai) 的 harness LLM(大语言模型)seam 通用多提供方适配器。一个插件实例拥有一份以路由为键的提供方 profile 字典;每个请求使用 `GenerateOptions.provider` 选择 profile,并从 pi-ai 已安装 catalog 中动态解析 `GenerateOptions.model`。
|
||||
基于 [`@earendil-works/pi-ai`](https://www.npmjs.com/package/@earendil-works/pi-ai) 的 harness LLM(大语言模型)seam 通用多提供方适配器。一个插件实例拥有一份以路由为键的提供方 profile 字典;每个请求使用 `GenerateOptions.provider` 选择 profile,并针对该路由已配置的 catalog 解析 `GenerateOptions.model`。点名了已安装 pi-ai 提供方的路由会继承其端点、协议格式与模型 catalog 作为默认值,并逐字段覆盖;pi-ai 未提供的路由则整体声明出来,因此接入 OpenAI 兼容网关、自建服务,或比已安装 catalog 更新的提供方,都属于配置而非改代码。
|
||||
|
||||
包(package)根入口导出 Cordis 插件契约与 `PiAiAdapter`;profile 解析、模型构造、回放转换和流转换保留在包内部。
|
||||
包(package)根入口导出 Cordis 插件契约、`PiAiAdapter` 与 `supportedProtocols()`;profile 解析、catalog 物化、提供方构造、回放转换和流转换保留在包内部。
|
||||
|
||||
## 配置
|
||||
|
||||
按提供方配置凭据与部署特定传输设置,并以提供方路由本身为键。优先使用 `apiKeyEnv`——按请求解析的凭据*引用*——而非字面 `apiKey`,让机密不进入该文件。**两者**都省略,才会把认证委托给 pi-ai 的提供方原生环境发现;已配置却解析不出任何值的引用则相反,会让请求以 `MISSING_CREDENTIAL` 失败,因为放行下去就会用环境里恰好持有的某个无关密钥完成认证。`baseURL` 只会覆盖所选 catalog 模型的端点,保留其 API 家族与兼容性元数据,因此仍支持 `https://proxy.example.com:8443` 等私有 proxy。
|
||||
按提供方配置凭据、模型 catalog 与部署特定传输设置,并以提供方路由本身为键。优先使用 `apiKeyEnv`——按请求解析的凭据*引用*——而非字面 `apiKey`,让机密不进入该文件。**两者**都省略,才会让该路由处于未认证状态;对已安装 catalog 路由而言,这意味着交给 pi-ai 的提供方原生环境发现。已配置却解析不出任何值的引用则相反,会让请求以 `MISSING_CREDENTIAL` 失败,因为放行下去就会用环境里恰好持有的某个无关密钥完成认证。一条凭据服务该路由下的全部模型。
|
||||
|
||||
```yaml
|
||||
- id: llm
|
||||
name: '@deepseek-ai/dsh-llm-pi-ai'
|
||||
config:
|
||||
providers:
|
||||
# Catalog route: endpoint, protocol, and models all come from pi-ai.
|
||||
openai:
|
||||
apiKeyEnv: OPENAI_API_KEY
|
||||
baseURL: https://proxy.example.com:8443
|
||||
@@ -26,36 +27,77 @@
|
||||
initialDelayMs: 500
|
||||
maxDelayMs: 10000
|
||||
jitterRatio: 0.1
|
||||
# Catalog route with its catalog narrowed to one model and that model's
|
||||
# capacity corrected; every unset field still comes from the catalog.
|
||||
anthropic:
|
||||
apiKeyEnv: ANTHROPIC_API_KEY
|
||||
streamIdleTimeoutMs: 300000
|
||||
openrouter:
|
||||
apiKeyEnv: OPENROUTER_API_KEY
|
||||
headers:
|
||||
X-Deployment: production
|
||||
models:
|
||||
- id: claude-sonnet-4-5
|
||||
contextWindow: 200000
|
||||
# Hand-declared route: pi-ai ships nothing under this key, so the profile
|
||||
# supplies the whole provider.
|
||||
acme-gateway:
|
||||
displayName: Acme Gateway
|
||||
apiKeyEnv: ACME_GATEWAY_API_KEY
|
||||
api: openai-completions
|
||||
baseURL: https://gateway.acme.example/v1
|
||||
models:
|
||||
- id: acme-large
|
||||
name: Acme Large
|
||||
contextWindow: 65536
|
||||
maxTokens: 4096
|
||||
```
|
||||
|
||||
每个字典键都必须存在于 pi-ai 已安装 catalog 中;字典形状使重复项无法表示,发布前的数组形状(每个 profile 携带 `provider` 字段)会加载失败并给出迁移指引。`providers` 也可以为空或整体省略:适配器将以**休眠**姿态挂载——零路由、模型选择器不多一条——一旦 `llm-pi-ai:` settings 分节提供了 profile 就即时注册路由,分节清空时随之撤销。无论是否休眠,插件都会在可配置提供方目录(`ctx.llm.listConfigurableProviders()`,settings 路径 `providers.<provider>`)中声明每个已安装 catalog 提供方,因此配置界面可以在任何路由存在之前就提供完整 catalog。哪些适配器存在归组合面;哪些提供方在运行可以完全交给用户的设置文档。向 `ctx.llm` 注册具有原子性:如果与另一适配器已拥有的任何提供方路由冲突,插件会加载失败,不注册剩余路由。模型 id 不是生命周期配置;未知模型会在发起任何提供方请求前以 `LlmError('UNKNOWN_MODEL')` 失败。
|
||||
字典形状使重复路由无法表示,发布前的数组形状(每个 profile 携带 `provider` 字段)会加载失败并给出迁移指引。`providers` 也可以为空或整体省略:适配器将以**休眠**姿态挂载——零路由、模型选择器不多一条——一旦 `llm-pi-ai:` settings 分节提供了 profile 就即时注册路由,分节清空时随之撤销。无论是否休眠,插件都会在可配置提供方目录(`ctx.llm.listConfigurableProviders()`,settings 路径 `providers.<provider>`)中声明每个已安装 catalog 提供方,并与当前 profile 声明的每条路由取并集,因此配置界面既能在任何路由存在之前就提供完整 catalog,也能寻址一条手工声明的路由。每个条目都带上 `declared`:pi-ai 在这个键下是否什么都没有。它跟随已安装 catalog 而非设置文档,因为收窄一个内置提供方的模型同样会存下 profile,而那条路由仍然是 pi-ai 认识的——只有适配器分得清两者,所以由目录直接给出答案,而不是留给界面去猜。哪些适配器存在归组合面;哪些提供方在运行可以完全交给用户的设置文档。向 `ctx.llm` 注册具有原子性:如果与另一适配器已拥有的任何提供方路由冲突,插件会加载失败,不注册剩余路由。模型 id 不是生命周期配置;路由未配置的模型会在发起任何提供方请求前以 `LlmError('UNKNOWN_MODEL')` 失败。
|
||||
|
||||
## Catalog 解析
|
||||
|
||||
profile 的 `models` 列表是*替换*该路由已安装 catalog,而不是扩充它;省略它(或留空)则原样服务该 catalog。每个条目都会从同 `id` 的已安装模型继承自身未设置的字段,因此把 catalog 路由收窄到两个模型、更正某个容量,或加入一个比已安装 catalog 更新的模型,都是一行编辑。只有 harness 会消费的字段可配置——`id`、`name`、`contextWindow` 与 `maxTokens`。定价与输入模态没有 harness 消费方,因此沿用已安装条目或直接缺席。推理则完全不按模型配置:一个孤立的能力布尔量会让 pi-ai 公布出没有 `thinkingLevelMap` 可供拼写的档位,而且没有任何列表端点会报告模型的推理协议,因此推理沿用已安装 catalog 条目或直接缺席。
|
||||
|
||||
条目与已安装 catalog 都没有给出尺寸的模型,会采用该路由的 `defaultContextWindow`(262,144)与 `defaultMaxTokens`(32,768),因此一份只公布 id 的列表同样能产出可服务的路由。两个回退值本质上都是猜测,这正是它们作为路由字段、供网关服务更小模型的部署一次性更正的原因,而不是埋在适配器里的常量;回退值只用于给模型定尺寸,绝不会变成每请求上限。
|
||||
|
||||
路由完全无法服务时解析仍会失败得响亮,并点名出问题的路由与模型:catalog 未提供的路由需要 `api`、`baseURL`,以及一个由唯一标识的模型组成的非空 `models` 列表。该解析在分节 schema 内部运行,因此无法服务的 profile 会在**写入之处**被拒绝——`settings.mutate` 以 `settings-rejected` 点名路由与模型——而不是先存下来、再悄悄让该 namespace 下每条路由失效。对于已经存下的、在此失败的分节,settings seam 会保留该 namespace 上一份可用值,因此这不会把部署卡死。`api` 接受 `supportedProtocols()` 中的协议,且仅在 catalog 无法提供协议时才需要:catalog 中不存在的模型会继承其同门模型一致同意的协议,因此向单协议 catalog 路由添加模型无需重述任何内容。
|
||||
|
||||
`baseURL` 设定该路由下每个模型的端点,因此仍支持 `https://proxy.example.com:8443` 等私有 proxy;省略它的 catalog 路由会保留每个 catalog 模型自己的端点。在 catalog 路由上点名 `api` 会把整条路由改指到该协议,这正是部署把某个提供方在 Responses 与 Chat Completions 之间迁移的方式。
|
||||
|
||||
`supportedProtocols()` 刻意窄于 pi-ai 的完整流式 API 集合:它只保留 profile 能用密钥、端点与标头**完整描述**的那些协议。Bedrock 要用 AWS 凭据与 region 做 SigV4 签名,Vertex 需要 project、location 与应用默认凭据,Azure 需要提供方环境外加 api-version,Codex 走 OAuth——提供它们只会交回一个无法完成认证的路由。catalog 路由仍可经自己的 provider 抵达这些协议;被拒绝的只有显式覆盖。
|
||||
|
||||
## 动态配置(settings + credentials)
|
||||
|
||||
适配器经由一个 thunk **每操作读取一次** profile,而非在构造期冻结。插件在可选的 `ctx.settings` seam 上用同一份 `Config` schema 注册 `llm-pi-ai` namespace,并以其 `cordis.yml` 条目为组合 `base`;由于 `providers` 是字典,base 与用户的 `llm-pi-ai:` settings 分节**按提供方**合并:用户可以新增路由、覆盖组合路由的单个字段,或把路由指向另一个 proxy,全部在下一次请求生效,无需重启。未挂载 settings 服务时,仅由 entry 配置驱动适配器,行为不变。
|
||||
|
||||
凭据按每次 stream 调用解析:非空的字面 `apiKey` 优先,其次经可选的 `ctx.credentials` seam 解析 `apiKeyEnv`(活跃环境之下的 `$DSH_HOME/.env`;未挂载 seam 时恰好读取该环境变量)。只有完全没有点名任何凭据的 profile——仅限这一种情况——才交给 pi-ai 的环境发现。路由集合与每条路由捕获的重试策略是注册级事实:两者任一变化时,插件都会原子地替换自己的注册(同一适配器实例,候选集合先经校验),因此某条路由若已被另一适配器占有,先前的路由会继续服务,而改回可用配置时注册会重新生效。提供方键的顺序绝不算作变化。存活 settings 快照若点名未知提供方(或违反任何其他 resolver 约束),则保留最后可用 profile 并记录失败;entry 配置本身仍会使插件加载失败。
|
||||
凭据按每次 stream 调用解析:非空的字面 `apiKey` 优先,其次经可选的 `ctx.credentials` seam 解析 `apiKeyEnv`(活跃环境之下的 `$DSH_HOME/.env`;未挂载 seam 时恰好读取该环境变量)。只有完全没有点名任何凭据的 profile——仅限这一种情况——才交给 pi-ai 的环境发现。每个密钥在使用前都会被去除首尾空白并校验格式——字面 `apiKey` 在 profile 解析时(插件加载,或下一次 settings 快照)校验,`apiKeyEnv` 解析出的值则在请求时校验——因此 HTTP 标头无法承载的值会在这一步被拒绝,而不是以语义不明的 `fetch` `TypeError` 形式浮现;请求时的拒绝会抛出 `LlmError('INVALID_CREDENTIAL')`,点名失败的路由与凭据引用,但绝不透露密钥的任何部分。路由集合与每条路由捕获的重试策略是注册级事实:两者任一变化时,插件都会原子地替换自己的注册(同一适配器实例,候选集合先经校验),因此某条路由若已被另一适配器占有,先前的路由会继续服务,而改回可用配置时注册会重新生效。提供方键的顺序绝不算作变化。本适配器无法服务的分节会在写入处被拒——注册的 `validate` 会解析整份 profile 集合,因此 `ctx.settings.mutate` 以 resolver 自身的错误拒绝(协议面将其报为 `settings-rejected`),什么都不会存储。已存储分节若因其他途径变得不可服务——比如外部编辑了 `settings.yaml`——则由 settings seam 保留该 namespace 最后可用的值并告警。entry 配置本身仍会使插件加载失败;而 llm 注册表拒绝的路由(已被另一适配器族占有的那种)会被记录下来,先前注册的路由继续服务。
|
||||
|
||||
适配器通过 `ctx.llm.listModels(provider)` 公开每个已配置提供方已安装的 pi-ai 模型。这是从 `getModels(provider)` 派生的提供方无关 selector 元数据;请求时解析仍会执行权威 catalog 查找,因此发现不会创建第二个模型注册表。`ctx.llm.resolveModelInfo(provider, model)` 会执行一次精确 descriptor 查找,并返回其身份、上下文窗口和可选思考级别,让权威元数据保留在拥有路由的适配器上,而非消费方。
|
||||
适配器通过 `ctx.llm.listModels(provider)` 公开每条已配置路由的模型。这是从请求路径所用的同一个 pi-ai `Models` 集合读取的提供方无关 selector 元数据,因此发现不会创建第二个模型注册表。`ctx.llm.resolveModelInfo(provider, model)` 会执行一次精确 descriptor 查找,并返回其身份、上下文窗口、已配置输出上限和可选思考级别,让权威元数据保留在拥有路由的适配器上,而非消费方。模型**已配置**的 `maxTokens` 会成为 seam 的 `defaultMaxTokens`,因此未点名输出上限的请求会携带部署选定的那一个;而从已安装 catalog 继承来的值是模型的输出**能力**,绝不会自行变成请求默认值。
|
||||
|
||||
`reasoning.efforts` 列表是 pi-ai 有序的 `getSupportedThinkingLevels(model)` 结果,不经筛选或规范化,其中包括 `off`,以及模型对 `xhigh` 或 `max` 的特定支持。Harness 将每个规范 pi-ai 级别公开为不透明 ID;提供方/模型在协议格式中的表示仍保留在 pi-ai 的 `thinkingLevelMap` 中。因此,不具备推理(reasoning)能力的模型也会公开 pi-ai 的 `off` 选项。配置 profile 的 `reasoning` 值(包括 `off`)在存在时是部署默认值;省略它会保留提供方默认值。每次请求的 `GenerateOptions.reasoningEffort` 优先;任何未出现在确切模型能力中的显式值都会在网络 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败,而不会被自动调整。pi-ai 的通用流选项通过省略 `reasoning` 表示 `off`。
|
||||
携带推理元数据的模型会公开 pi-ai 有序的 `getSupportedThinkingLevels(model)` 结果,不经筛选或规范化,其中包括 `off`,以及模型对 `xhigh` 或 `max` 的特定支持。Harness 将每个规范 pi-ai 级别公开为不透明 ID;提供方/模型在协议格式中的表示仍保留在 pi-ai 的 `thinkingLevelMap` 中。
|
||||
|
||||
受支持的 profile 字段是 `apiKey`、`apiKeyEnv`、`baseURL`、`headers`、`reasoning`、`thinkingBudgets`、`cacheRetention`、`transport`、`timeoutMs`、`websocketConnectTimeoutMs`、`streamIdleTimeoutMs` 和 `retryPolicy`。每个 profile 的可选重试策略都会与该提供方路由一同捕获;省略时使用有界的常规默认值。流空闲间隔必须是正的有限 Node 定时器延迟,默认为五分钟,且只覆盖未完成提供方读取,不包括消费方思考时间。若已配置标头中有同名项,则以 Harness 应用归因为准。
|
||||
**没有**这份元数据的模型——每一个手工声明的模型,以及 pi-ai 标记为不具备推理能力的 catalog 模型——完全不公开 `reasoning`。pi-ai 会把这类模型报告为只支持 `off` 一档,但 `off` 会被翻译成*省略* reasoning 选项,而那与「不点名任何档位」产出的请求逐字节相同:选它关不掉任何东西,于是自身默认就在思考的提供方,会在界面显示 `off` 被选中的同时继续思考。把该能力报告为不可用,界面就只剩提供方默认这一项,不会再出现自相矛盾的控件。配置 profile 的 `reasoning` 值(包括 `off`)在存在时是部署默认值;省略它会保留提供方默认值。每次请求的 `GenerateOptions.reasoningEffort` 优先;未出现在确切模型能力中的档位会让**请求**在网络 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败,而不会被自动调整。**描述**一个模型则从不这样失败:同一提供方下各模型接受的档位并不一致,因此 `resolveModel` 对该模型拿不下的 profile 档位报告为「没有默认值」,而不是抛错。在那里抛错会让整个提供方从任何基于它构建的模型目录中消失——一个配错的 profile 字段连支持该档位的模型也一并藏起来——所以坏配置暴露在被执行处,而不是被描述处。pi-ai 的通用流选项通过省略 `reasoning` 表示 `off`。
|
||||
|
||||
受支持的 profile 字段是 `apiKey`、`apiKeyEnv`、`displayName`、`api`、`baseURL`、`models`、`defaultContextWindow`、`defaultMaxTokens`、`headers`、`reasoning`、`thinkingBudgets`、`cacheRetention`、`transport`、`timeoutMs`、`websocketConnectTimeoutMs`、`streamIdleTimeoutMs` 和 `retryPolicy`。每个 profile 的可选重试策略都会与该提供方路由一同捕获;省略时使用有界的常规默认值。流空闲间隔必须是正的有限 Node 定时器延迟,默认为五分钟,且只覆盖未完成提供方读取,不包括消费方思考时间。若已配置标头中有同名项,则以 Harness 应用归因为准。
|
||||
|
||||
适配器强制 pi-ai SDK `maxRetries` 为零,因此一次 `stream()` 调用只会发起一次提供方请求。已移除 profile 字段 `maxRetries` 和 `maxRetryDelayMs` 会使加载失败,而不是静默倍增或隐藏单独组合的 agent(智能体)级重试预算。空闲超时会 abort SDK 的稳定请求信号,并以 `TIMEOUT` 呈现;较早的调用方 abort 仍为 `ABORTED`。
|
||||
|
||||
图片请求会在请求分发时解析可选的 `ctx.attachments` 服务,因此 Cordis 插件加载顺序不会固化附件可用性。图片检测与转换会递归遍历嵌套的 `tool-result` 内容,因此嵌套图片既不会被展平,也不会被跳过。当该服务或所选模型的图片能力不存在时,视觉请求仍会明确以 `UNSUPPORTED_CONTENT` 失败。
|
||||
## 端点询问
|
||||
|
||||
插件提供 `ctx.llm.registerModelDiscovery('llm-pi-ai', …)`,用来回答「这个提供方能服务哪些模型?」——针对配置界面正在编辑或起草的路由。它刻意**不是** catalog 刷新:什么都不存储,回复是界面供用户采纳的候选。`settings.yaml` 始终是唯一决定路由服务什么的东西。
|
||||
|
||||
点名了**已安装 catalog 所提供路由**的请求,直接由该 catalog 作答,完全不联网:pi-ai 的注册表才是它自家提供方的权威列表,且携带列表端点不会公布的上下文窗口与输出上限。这类路由根本不需要 `baseURL`。只有 catalog 未描述的路由——网关、自建服务——才会经协议层询问;若它也没给端点,则会被告知去设置一个或手工填写模型。
|
||||
|
||||
草稿携带的是用户当下键入的凭据(如果有);已经存好凭据的路由,在配置界面上只呈现一个脱敏描述符,因此询问会自行取用该路由的凭据——解析方式与向它发请求时完全一致,先 `apiKey` 后 `apiKeyEnv`——而不是不带认证发出去、再把端点的 401 报成密钥不对。键入的密钥优先,因为那正是被测试的那一把。解析只发生在真正要联网的路径上,因此 catalog 路由作答时完全不会触碰凭据。用户提供或已存储的探测密钥也会经过同样的去除空白与格式校验:HTTP 标头无法承载的值会被立即以 `LlmError('INVALID_CREDENTIAL')` 拒绝,而不会传到 `fetch`——否则会呈现为一个和端点不可达难以区分的、语义不明的 `ByteString` 失败。
|
||||
|
||||
询问只读 `openai-completions` 与 `openai-responses`,它们「`GET /models` + bearer 认证」的形状是网关、自建服务与官方端点三方一致认可的那一种。Azure 尽管出身 OpenAI 也被排除——它用 `api-key` 标头认证并要求 `api-version` 查询参数——Codex 则走 OAuth;其余协议一律以 `DISCOVERY_UNSUPPORTED` 回答,让界面回退到手工填写,而不是把认证失败报成一个没有模型的提供方。`baseURL` 按前缀而非待解析 URL 处理,因此 `https://gateway.example/openai/v1` 这类部署路径会保留其路径段。
|
||||
|
||||
多数列表只公布 id;`context_window`/`context_length` 与 `max_output_tokens`/`max_tokens` 在网关提供时会被读取,没有可用 id 的条目会被跳过而不是让整份列表失败,其余仍由采纳方补齐。回复在四兆字节上限下读取,且上限落在实际收到的字节上——端点是用户自己填的 URL,因此会先看声明长度,但绝不把它当作边界。端点不可达、凭据被拒、响应非 JSON、以及响应没有 `data` 数组,都会以 `DISCOVERY_FAILED` 失败,消息点名端点;仅当 401 或 403 时才点名凭据。读取响应体期间被取消会呈现为 `ABORTED`,与请求发出之前被取消一致。
|
||||
|
||||
## 提供方/模型路由与回放
|
||||
|
||||
所选 pi-ai catalog descriptor 提供协议实现。这包括原生 API 差异,例如 descriptor 使用 Responses API 而非 Chat Completions 的 OpenAI 模型;harness 适配器不会按模型名称硬编码端点选择。
|
||||
每次解析产出一份**不可变**快照——profiles 加上一个持有各路由所建 `Provider` 的 `createModels()` 集合——每个操作都在自己第一个 `await` 之前整体捕获一份快照。配置变化会构造**新**集合,而不是改动正在被使用的那个:`Models.streamSimple()` 是惰性的,它在流首次被消费时才解析 provider,而那已在 credential await 之后,因此改动共享集合会让一个在旧配置下开始的请求在新配置下结束,或者撞上一个已不存在的 provider。这正是 seam 的每步调用冻结(`llm.prepareCall()`)能贯通到底的原因——回复途中切换模型会在下一步生效,绝不会影响在途的那一步。请求经 `Models.streamSimple()` 抵达提供方。保持 catalog 协议不变的 catalog 路由会**复用**已安装提供方,只替换其模型列表,因为该提供方持有本包无法重建的 API 实现——Bedrock 经由独立入口加载其 Smithy 模块——从零件重建会静默收窄可用提供方的范围。其余路由都由 `createProvider()` 基于 `supportedProtocols()` 背后的协议表构造,表中条目正是 pi-ai 自己的提供方工厂所用的同一批 factory。
|
||||
|
||||
凭据绝不进入该集合。harness 在请求抵达 pi-ai 之前经自身 seam 解析路由密钥,并作为请求的 `apiKey` 选项传入,而 pi-ai 将其视为优先级最高的 auth 覆盖;因此 `Models` 不持有任何凭据存储,harness 也保住了自己失败得响亮的引用语义。没有点名任何凭据的路由会解析为「已配置但无密钥」,把该要求留给协议——那才是它真正所在的位置。
|
||||
|
||||
所选模型 descriptor 提供协议实现。这包括原生 API 差异,例如 descriptor 使用 Responses API 而非 Chat Completions 的 OpenAI 模型;harness 适配器不会按模型名称硬编码端点选择。
|
||||
|
||||
成功的 assistant 响应会在自身持久提供方/模型溯源旁存储经版本化的无损 JSON 回放状态。请求时,`LlmService` 只有在历史提供方路由与目标提供方路由当前由同一个 `PiAiAdapter` 实例拥有时,才会传递回放状态。即使目标提供方或模型改变,适配器也会验证状态并恢复 pi-ai 响应 id 与提供方 signature;随后由 pi-ai 判定目标 API 可以复用哪些元数据。没有回放状态的历史会被转换为外来的、与提供方无关的内容,绝不伪装为原生 pi-ai 响应。
|
||||
|
||||
@@ -77,10 +119,6 @@
|
||||
|
||||
pi-ai 会安装多个提供方 SDK,并延迟加载 catalog 模型所选的 SDK。该可选适配器包将依赖体量隔离在自身范围内。
|
||||
|
||||
## 测试
|
||||
|
||||
单元测试使用重定向到本地 mock 服务器的 pi-ai catalog 模型,覆盖提供方/profile 路由、每次适配器调用只发起一个协议请求、idle-timeout 响应终止、调用方 abort、原生 API 选择、端点覆盖、归因、转换、回放状态验证,以及一个适配器实例内的跨提供方/模型回放。`tests/dynamic-config.spec.ts` 驱动真实的 settings-local 与 credentials-local provider:settings 里新生的路由实时完成注册,并在用户层重置时随之移除,`apiKeyEnv` 凭据在两次请求之间轮换,点名未知提供方的快照则保留最后可用 profile。`tests/loader-composition.spec.ts` 从仅测试用的 `cordis.yml` 出发,经真实 Loader 拉起休眠姿态,并从磁盘上的一次 `settings.yaml` 编辑注册出它的路由。真实 API 覆盖仍需 key 才会启用,并通过 `pnpm run test:e2e` 运行。
|
||||
|
||||
## 模型体验
|
||||
|
||||
### 通过 pi-ai 发起的提供方请求
|
||||
@@ -115,7 +153,9 @@ pi-ai 事件会变为 harness 推理、文本、工具调用、usage 与 finish
|
||||
|
||||
- **settings 能新增或覆盖路由,但不能移除组合路由**:用户层合并在组合 `base` 之上,因此删除 `cordis.yml` 提供的提供方属于组合变更;对该 namespace 执行 `replace` 只会重置用户层。
|
||||
- **`headers` 可能承载一条脱敏器看不见的凭据**:profile 的 `headers` 是纯字符串字典,因此设在其中的 `Authorization` 或 `api-key` 会被脱敏后的 `describe()` 原样返回,并被任何配置 UI 渲染出来。请把凭据存为 `apiKeyEnv` 引用;把该字典整体改为只写与其余[协议边界工作](../llm/README.md#known-limitations-and-deferred-work)一并暂缓。
|
||||
- **必须属于 catalog**:已安装 pi-ai catalog 中不存在的自定义模型 id 会以 `UNKNOWN_MODEL` 失败,即使提供方 profile 配置了自定义端点。
|
||||
- **路由的 catalog 不会自我刷新**:catalog 就是 `settings.yaml` 所写的内容,因此模型列表的新鲜度只到最近一次编辑为止。这里没有任何环节会去问提供方它服务哪些模型;路由要多一个模型,得有人写进去。
|
||||
- **每条路由只有一种协议格式**:`api` 作用于整条路由,因此混合协议的 catalog 路由(跨 Responses 与 Chat Completions 的 OpenAI 式 catalog)无法承载另一种协议的模型,向这类路由添加它未描述的模型必须点名 `api` 并把全部模型一起迁过去。把该提供方拆成两个路由键是变通办法。
|
||||
- **未认证路由取决于其协议**:不点名凭据会让路由解析为「已配置但无密钥」,但 pi-ai 的 OpenAI 兼容实现仍要求 API key 或 `Authorization` 标头,因此无鉴权的本地服务需要一个占位 `apiKey`,或在 `headers` 中给出 `Authorization` 条目。
|
||||
- **不支持 `GenerateOptions.stop`**:pi-ai 的通用流选项无法保证所有提供方都支持 stop sequence,因此适配器会拒绝该字段。
|
||||
- **历史中的 `system` 消息使用 pi-ai 通用上下文转换**:提供方特定位置由 pi-ai 决定,而非由 harness 拥有的协议覆盖决定。
|
||||
- **无法获取提供方 HTTP 状态**:pi-ai 错误事件不会在所有提供方上公开稳定 HTTP 状态;失败只公开稳定 harness 错误 code。
|
||||
|
||||
@@ -21,9 +21,7 @@
|
||||
"files": [
|
||||
"lib/index.js",
|
||||
"lib/invariant.js",
|
||||
"lib/types/**/*.d.ts",
|
||||
"lib/types/**/*.d.ts.map",
|
||||
"src"
|
||||
"lib/types/**/*.d.ts"
|
||||
],
|
||||
"license": "BSD-3-Clause",
|
||||
"peerDependencies": {
|
||||
|
||||
@@ -1,21 +1,36 @@
|
||||
/**
|
||||
* Generic pi-ai-backed implementation of the Harness LLM seam.
|
||||
*
|
||||
* Each resolution produces one **immutable** snapshot — the profiles plus a
|
||||
* `Models` collection holding the `Provider` each route built — and an
|
||||
* operation captures a whole snapshot before its first `await`. A
|
||||
* configuration change builds a *new* collection rather than mutating the one
|
||||
* in use, because `Models.streamSimple()` is lazy: it resolves the provider
|
||||
* when the stream is first consumed, which is after the credential await, so a
|
||||
* mutated collection would let a request that started under one configuration
|
||||
* finish under another — or fail with a provider that no longer exists. This is
|
||||
* what makes the seam's per-step call freeze (`llm.prepareCall()`) hold all the
|
||||
* way down: switching models mid-reply takes effect on the next step, never
|
||||
* inside the one in flight.
|
||||
*
|
||||
* Credentials stay outside that collection. The harness resolves a route's key
|
||||
* through its own seam and passes it as the request's `apiKey` option, which
|
||||
* pi-ai treats as the highest-priority auth override — so `Models` never holds
|
||||
* a credential store and the harness keeps its fail-loud reference semantics.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/adapter
|
||||
*/
|
||||
|
||||
import { streamSimple } from '@earendil-works/pi-ai/compat'
|
||||
import { getBuiltinModels } from '@earendil-works/pi-ai/providers/all'
|
||||
import type { BuiltinProvider } from '@earendil-works/pi-ai/providers/all'
|
||||
import { getSupportedThinkingLevels } from '@earendil-works/pi-ai'
|
||||
import { createModels, getSupportedThinkingLevels } from '@earendil-works/pi-ai'
|
||||
import type {
|
||||
Api,
|
||||
Model,
|
||||
Models,
|
||||
ModelThinkingLevel,
|
||||
MutableModels,
|
||||
SimpleStreamOptions,
|
||||
ThinkingLevel,
|
||||
} from '@earendil-works/pi-ai'
|
||||
import type { AttachmentStore } from '@deepseek-ai/dsh-attachment'
|
||||
import {
|
||||
attributionHeaders,
|
||||
contentHasImage,
|
||||
@@ -26,47 +41,43 @@ import {
|
||||
import type {
|
||||
GenerateOptions,
|
||||
LlmModelInfo,
|
||||
LlmProviderInfo,
|
||||
LlmResolvedModelInfo,
|
||||
ReasoningEffortId as ReasoningEffortIdType,
|
||||
ResolvedRetryPolicy,
|
||||
StreamChunk,
|
||||
} from '@deepseek-ai/dsh-llm'
|
||||
import type { AttachmentStore } from '@deepseek-ai/dsh-attachment'
|
||||
import { idleWatchdog, timeoutOf } from '@deepseek-ai/dsh-timeout'
|
||||
import type { ResolvedPiAiProviderProfile } from './config.ts'
|
||||
import { toPiContext } from './context.ts'
|
||||
import { toStreamChunks } from './stream.ts'
|
||||
|
||||
/** Constructor options for {@link PiAiAdapter}: the request-time resolution seams the plugin owns. */
|
||||
/** One resolution's frozen view: the profiles and the collection built from them. */
|
||||
interface PiAiSnapshot {
|
||||
/** The resolved profiles this collection was built from, used as its identity. */
|
||||
profiles: ReadonlyMap<string, ResolvedPiAiProviderProfile>
|
||||
/** Providers for exactly those profiles; never mutated once published. */
|
||||
models: Models
|
||||
}
|
||||
|
||||
/** Constructor options for {@link PiAiAdapter}: the two resolution seams the plugin owns. */
|
||||
export interface PiAiAdapterOptions {
|
||||
/** Current validated profiles by provider route; called once per operation. */
|
||||
profiles: () => ReadonlyMap<string, ResolvedPiAiProviderProfile>
|
||||
/**
|
||||
* Resolve the credential for one already-resolved profile; called once per
|
||||
* stream call and frozen for that call. `undefined` defers to pi-ai's
|
||||
* provider-native ambient discovery, which the plugin allows only for a
|
||||
* profile naming no credential at all; a named reference that misses throws
|
||||
* `LlmError` `MISSING_CREDENTIAL` rather than falling back.
|
||||
* stream call and frozen for that call. `undefined` defers to the route's own
|
||||
* pi-ai auth, which for an installed catalog route is its provider-native
|
||||
* ambient discovery; the plugin allows that only for a profile naming no
|
||||
* credential at all, because a named reference that misses throws `LlmError`
|
||||
* `MISSING_CREDENTIAL` rather than falling back.
|
||||
*/
|
||||
resolveApiKey: (provider: string, profile: ResolvedPiAiProviderProfile) => Promise<string | undefined>
|
||||
/** Resolve durable image storage at request time so plugin load order does not become capability state. */
|
||||
/** Resolve the optional durable attachment service at request time. */
|
||||
resolveAttachments?: () => AttachmentStore | undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a catalog model dynamically and apply only the configured endpoint
|
||||
* override, preserving the catalog's API/capability/compatibility metadata.
|
||||
*/
|
||||
function resolvePiModel(
|
||||
profile: ResolvedPiAiProviderProfile,
|
||||
modelId: string,
|
||||
): Model<Api> {
|
||||
const model = getBuiltinModels(profile.provider as BuiltinProvider).find(candidate => candidate.id === modelId) as Model<Api> | undefined
|
||||
if (model === undefined) {
|
||||
throw new LlmError(`pi-ai provider "${profile.provider}" has no catalog model "${modelId}"`, 'UNKNOWN_MODEL')
|
||||
}
|
||||
return profile.baseURL === undefined ? model : { ...model, baseUrl: profile.baseURL }
|
||||
}
|
||||
|
||||
/** Copy profile stream knobs into pi-ai's common option vocabulary. */
|
||||
function profileOptions(
|
||||
profile: ResolvedPiAiProviderProfile,
|
||||
@@ -87,6 +98,29 @@ function profileOptions(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile default this exact model can actually take, for DESCRIBING it.
|
||||
* A configured level the model does not support yields none rather than
|
||||
* throwing: `resolveModel` builds the model catalog, and a catalog that fails
|
||||
* takes its whole provider out of every picker — so one mis-set profile field
|
||||
* would hide every model on the route, including the ones that support the
|
||||
* level. The request path still refuses, which is where a bad configuration
|
||||
* belongs: describing what a model can do must not fail because a deployment
|
||||
* asked it for something it cannot.
|
||||
* @param model - the resolved model descriptor.
|
||||
* @param effort - the profile's configured level, if any.
|
||||
* @returns the level when this model supports it, otherwise undefined.
|
||||
*/
|
||||
function describableReasoningLevel(
|
||||
model: Model<Api>,
|
||||
effort: ReasoningEffortIdType | ModelThinkingLevel | undefined,
|
||||
): ModelThinkingLevel | undefined {
|
||||
if (effort === undefined) return undefined
|
||||
return getSupportedThinkingLevels(model).some(level => level === effort)
|
||||
? effort as ModelThinkingLevel
|
||||
: undefined
|
||||
}
|
||||
|
||||
/** Validate an explicit Harness/profile effort without invoking pi-ai's clamp. */
|
||||
function resolveReasoningLevel(
|
||||
model: Model<Api>,
|
||||
@@ -101,6 +135,39 @@ function resolveReasoningLevel(
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Selectable reasoning efforts for one model, or nothing at all.
|
||||
*
|
||||
* A model that carries no reasoning metadata — every hand-declared one, and
|
||||
* every catalog model pi-ai marks as non-reasoning — is reported by pi-ai as
|
||||
* supporting the single level `off`. Passing that through would offer a control
|
||||
* that cannot do what it says: `off` is translated to *omitting* the reasoning
|
||||
* option, which for such a model is byte-for-byte the same request as naming no
|
||||
* effort — so a provider whose own default is to think would keep thinking with
|
||||
* `off` selected. Omitting `reasoning` entirely is the seam's way of saying the
|
||||
* capability is unavailable, which leaves the surface offering only the
|
||||
* provider's default.
|
||||
* @param model - the resolved model descriptor.
|
||||
* @param defaultLevel - the profile's configured effort, already validated.
|
||||
* @returns the `reasoning` field, or an empty object when none can be offered.
|
||||
*/
|
||||
function reasoningInfo(
|
||||
model: Model<Api>,
|
||||
defaultLevel: ModelThinkingLevel | undefined,
|
||||
): Pick<LlmResolvedModelInfo, 'reasoning'> | Record<string, never> {
|
||||
if (!model.reasoning) return {}
|
||||
const levels = getSupportedThinkingLevels(model)
|
||||
return {
|
||||
reasoning: {
|
||||
efforts: levels.map(level => ({
|
||||
id: ReasoningEffortId(level),
|
||||
name: `${level.charAt(0).toUpperCase()}${level.slice(1)}`,
|
||||
})),
|
||||
...defaultLevel === undefined ? {} : { defaultEffort: ReasoningEffortId(defaultLevel) },
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/** Merge deployment headers while removing case-insensitive attribution collisions. */
|
||||
function requestHeaders(headers: Readonly<Record<string, string>> | undefined): Record<string, string> {
|
||||
const attribution = attributionHeaders()
|
||||
@@ -112,29 +179,73 @@ function requestHeaders(headers: Readonly<Record<string, string>> | undefined):
|
||||
}
|
||||
|
||||
/**
|
||||
* pi-ai-backed multi-provider adapter. Model descriptors are resolved for each
|
||||
* request, so models need not be registered during the Cordis lifecycle.
|
||||
* pi-ai-backed multi-provider adapter. Each operation reads the current
|
||||
* profiles, so a configuration change reaches the next request without a
|
||||
* restart; model descriptors come from the collection those profiles built.
|
||||
*/
|
||||
export class PiAiAdapter extends LlmAdapter {
|
||||
private snapshot: PiAiSnapshot | undefined
|
||||
|
||||
constructor(private readonly config: PiAiAdapterOptions) {
|
||||
super()
|
||||
}
|
||||
|
||||
/**
|
||||
* The snapshot for the current profiles. Resolution memoizes its result, so
|
||||
* an unchanged configuration is recognized by identity; a changed one gets a
|
||||
* brand-new collection, leaving any snapshot an operation already captured
|
||||
* untouched for as long as that operation holds it.
|
||||
*/
|
||||
private current(): PiAiSnapshot {
|
||||
const profiles = this.config.profiles()
|
||||
if (this.snapshot?.profiles === profiles) return this.snapshot
|
||||
const models: MutableModels = createModels()
|
||||
for (const profile of profiles.values()) models.setProvider(profile.piProvider)
|
||||
this.snapshot = { profiles, models }
|
||||
return this.snapshot
|
||||
}
|
||||
|
||||
/** The profile for one route within one snapshot, or the not-owned failure. */
|
||||
private profileOf(snapshot: PiAiSnapshot, provider: string): ResolvedPiAiProviderProfile {
|
||||
const profile = snapshot.profiles.get(provider)
|
||||
if (profile === undefined) {
|
||||
throw new LlmError(`pi-ai adapter does not own provider "${provider}"`, 'NO_ADAPTER')
|
||||
}
|
||||
return profile
|
||||
}
|
||||
|
||||
/** The configured descriptor for one exact route/model pair within one snapshot. */
|
||||
private modelOf(snapshot: PiAiSnapshot, provider: string, model: string): Model<Api> {
|
||||
this.profileOf(snapshot, provider)
|
||||
const resolved = snapshot.models.getModel(provider, model)
|
||||
if (resolved === undefined) {
|
||||
throw new LlmError(`pi-ai provider "${provider}" has no configured model "${model}"`, 'UNKNOWN_MODEL')
|
||||
}
|
||||
return resolved
|
||||
}
|
||||
|
||||
override providerInfo(provider: string): LlmProviderInfo {
|
||||
// The configured name, not the route key: `displayName` exists so a
|
||||
// deployment can label a route, and a label only the configuration surface
|
||||
// reads would leave every selector showing the raw key.
|
||||
return { id: provider, name: this.current().profiles.get(provider)?.displayName ?? provider }
|
||||
}
|
||||
|
||||
override providerRetryPolicy(provider: string): ResolvedRetryPolicy | undefined {
|
||||
return this.config.profiles().get(provider)?.retryPolicy
|
||||
return this.current().profiles.get(provider)?.retryPolicy
|
||||
}
|
||||
|
||||
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
|
||||
const profile = this.config.profiles().get(provider)
|
||||
if (profile === undefined) {
|
||||
return Promise.reject(new LlmError(`pi-ai adapter does not own provider "${provider}"`, 'NO_ADAPTER'))
|
||||
}
|
||||
return Promise.resolve(getBuiltinModels(profile.provider as BuiltinProvider).map(model => ({
|
||||
provider,
|
||||
id: model.id,
|
||||
name: model.name,
|
||||
inputModalities: [...model.input],
|
||||
})))
|
||||
return Promise.resolve().then(() => {
|
||||
const snapshot = this.current()
|
||||
this.profileOf(snapshot, provider)
|
||||
return snapshot.models.getModels(provider).map(model => ({
|
||||
provider,
|
||||
id: model.id,
|
||||
name: model.name,
|
||||
inputModalities: [...model.input],
|
||||
}))
|
||||
})
|
||||
}
|
||||
|
||||
override resolveModel(
|
||||
@@ -142,32 +253,22 @@ export class PiAiAdapter extends LlmAdapter {
|
||||
model: string,
|
||||
_signal?: AbortSignal,
|
||||
): Promise<LlmResolvedModelInfo> {
|
||||
const profile = this.config.profiles().get(provider)
|
||||
if (profile === undefined) {
|
||||
return Promise.reject(new LlmError(
|
||||
`pi-ai adapter does not own provider "${provider}"`,
|
||||
'NO_ADAPTER',
|
||||
))
|
||||
}
|
||||
return Promise.resolve().then(() => {
|
||||
const resolvedModel = resolvePiModel(profile, model)
|
||||
const levels = getSupportedThinkingLevels(resolvedModel)
|
||||
const defaultLevel = resolveReasoningLevel(resolvedModel, profile.reasoning)
|
||||
const snapshot = this.current()
|
||||
const profile = this.profileOf(snapshot, provider)
|
||||
const resolvedModel = this.modelOf(snapshot, provider, model)
|
||||
const defaultLevel = describableReasoningLevel(resolvedModel, profile.reasoning)
|
||||
// Only a cap the deployment configured is a request default; the
|
||||
// catalog's `maxTokens` sizes the model and stops there.
|
||||
const configuredMaxTokens = profile.configuredMaxTokens.get(model)
|
||||
return {
|
||||
provider,
|
||||
id: model,
|
||||
name: resolvedModel.name,
|
||||
inputModalities: [...resolvedModel.input],
|
||||
context: { contextWindow: resolvedModel.contextWindow },
|
||||
reasoning: {
|
||||
efforts: levels.map(level => ({
|
||||
id: ReasoningEffortId(level),
|
||||
name: `${level.charAt(0).toUpperCase()}${level.slice(1)}`,
|
||||
})),
|
||||
...defaultLevel === undefined
|
||||
? {}
|
||||
: { defaultEffort: ReasoningEffortId(defaultLevel) },
|
||||
},
|
||||
...configuredMaxTokens === undefined ? {} : { defaultMaxTokens: configuredMaxTokens },
|
||||
...reasoningInfo(resolvedModel, defaultLevel),
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -176,14 +277,14 @@ export class PiAiAdapter extends LlmAdapter {
|
||||
if (options.stop !== undefined) {
|
||||
throw new LlmError('llm-pi-ai does not support GenerateOptions.stop', 'UNSUPPORTED_OPTION')
|
||||
}
|
||||
// One resolution per stream call: the profile snapshot and the credential
|
||||
// freeze here and hold for this whole request, so an in-flight stream
|
||||
// never observes a configuration change and the next call re-resolves.
|
||||
const profile = this.config.profiles().get(options.provider)
|
||||
if (profile === undefined) {
|
||||
throw new LlmError(`pi-ai adapter does not own provider "${options.provider}"`, 'NO_ADAPTER')
|
||||
}
|
||||
const model = resolvePiModel(profile, options.model)
|
||||
// One capture per stream call, taken before any await: the profile, the
|
||||
// model descriptor, and the collection all come from the same immutable
|
||||
// snapshot, and the credential freezes with them. A configuration change
|
||||
// mid-request builds a separate snapshot, so this request finishes under
|
||||
// the one it started with and the next call picks up the new one.
|
||||
const snapshot = this.current()
|
||||
const profile = this.profileOf(snapshot, options.provider)
|
||||
const model = this.modelOf(snapshot, options.provider, options.model)
|
||||
const reasoning = resolveReasoningLevel(
|
||||
model,
|
||||
options.reasoningEffort ?? profile.reasoning,
|
||||
@@ -209,7 +310,7 @@ export class PiAiAdapter extends LlmAdapter {
|
||||
const context = attachments === undefined
|
||||
? toPiContext(options)
|
||||
: await toPiContext(options, attachments)
|
||||
const events = streamSimple(model, context, {
|
||||
const events = snapshot.models.streamSimple(model, context, {
|
||||
...profileOptions(profile, reasoning, apiKey),
|
||||
...options.temperature === undefined ? {} : { temperature: options.temperature },
|
||||
...options.maxTokens === undefined ? {} : { maxTokens: options.maxTokens },
|
||||
|
||||
223
packages/llm/llm-pi-ai/src/catalog.ts
Normal file
223
packages/llm/llm-pi-ai/src/catalog.ts
Normal file
@@ -0,0 +1,223 @@
|
||||
/**
|
||||
* Materialization of one provider route's model catalog. The installed pi-ai
|
||||
* catalog supplies defaults keyed by model id, and a profile's own model
|
||||
* entries override them field by field, so a route naming a catalog provider
|
||||
* stays configuration-free while a route pi-ai has never heard of is fully
|
||||
* describable from `settings.yaml`.
|
||||
*
|
||||
* Every pi-ai `Model` field the harness cannot default is required here rather
|
||||
* than at request time: an unserviceable route fails while its configuration is
|
||||
* being resolved, which is the earliest point that can name the offending key.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/catalog
|
||||
*/
|
||||
|
||||
import { builtinProviders, getBuiltinModels, getBuiltinProviders } from '@earendil-works/pi-ai/providers/all'
|
||||
import type { BuiltinProvider } from '@earendil-works/pi-ai/providers/all'
|
||||
import type { Api, Model, ModelCost, Provider } from '@earendil-works/pi-ai'
|
||||
|
||||
/**
|
||||
* Pricing for a model the installed catalog does not describe. The harness
|
||||
* never reads pi-ai's cost metadata — `replay.ts` zeroes it and no consumer
|
||||
* reports spend — so this is the absence of a fact, not a configurable rate.
|
||||
*/
|
||||
const NO_COST: ModelCost = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }
|
||||
|
||||
/**
|
||||
* Input modalities for a model the installed catalog does not describe. The
|
||||
* request converter keeps only text blocks, so text is the adapter's actual
|
||||
* capability rather than a deployment choice.
|
||||
*/
|
||||
const TEXT_ONLY: Model<Api>['input'] = ['text']
|
||||
|
||||
let providerIndex: Map<string, Provider> | undefined
|
||||
|
||||
/**
|
||||
* Installed catalog providers by id, constructed once. Each entry owns the API
|
||||
* implementations for its own models, which is why a catalog route reuses this
|
||||
* provider instead of being rebuilt from parts.
|
||||
* @returns the catalog provider index.
|
||||
*/
|
||||
function catalogProviders(): Map<string, Provider> {
|
||||
providerIndex ??= new Map(builtinProviders().map(provider => [provider.id, provider]))
|
||||
return providerIndex
|
||||
}
|
||||
|
||||
/**
|
||||
* The installed catalog provider for one route, when pi-ai ships one.
|
||||
* @param provider - provider route key.
|
||||
* @returns the catalog provider, or `undefined` for a route pi-ai does not ship.
|
||||
*/
|
||||
export function catalogProvider(provider: string): Provider | undefined {
|
||||
return catalogProviders().get(provider)
|
||||
}
|
||||
|
||||
/**
|
||||
* Every provider route the installed pi-ai catalog ships.
|
||||
* @returns the catalog provider ids.
|
||||
*/
|
||||
export function catalogProviderIds(): readonly string[] {
|
||||
return getBuiltinProviders()
|
||||
}
|
||||
|
||||
/**
|
||||
* The installed catalog models for one route, indexed by model id.
|
||||
* @param provider - provider route key.
|
||||
* @returns catalog models by id; empty for a route pi-ai does not ship.
|
||||
*/
|
||||
export function catalogModels(provider: string): Map<string, Model<Api>> {
|
||||
if (!catalogProviders().has(provider)) return new Map()
|
||||
const models = getBuiltinModels(provider as BuiltinProvider) as Model<Api>[]
|
||||
return new Map(models.map(model => [model.id, model]))
|
||||
}
|
||||
|
||||
/** One configured model entry: an id plus the catalog fields it overrides. */
|
||||
export interface PiAiModelProfile {
|
||||
/** Model id sent to the provider and accepted by {@link GenerateOptions.model}. */
|
||||
id: string
|
||||
/** Display name for selectors; defaults to the catalog name, then the id. */
|
||||
name?: string
|
||||
/** Maximum combined request and response context in tokens. */
|
||||
contextWindow?: number
|
||||
/**
|
||||
* Maximum output tokens. Configuring one also makes it this model's
|
||||
* per-request default; a value inherited from the installed catalog, or the
|
||||
* route's fallback, is the model's capability and never becomes a request
|
||||
* default on its own.
|
||||
*/
|
||||
maxTokens?: number
|
||||
}
|
||||
|
||||
/** The route-level facts model materialization reads. */
|
||||
export interface RouteCatalogRequest {
|
||||
/** Provider route key, stamped onto every materialized model. */
|
||||
provider: string
|
||||
/** Wire protocol override; absent defers to each catalog model's own API. */
|
||||
api?: string
|
||||
/** Endpoint override; absent defers to the catalog model, then the catalog provider. */
|
||||
baseURL?: string
|
||||
/** Configured catalog; absent means the whole installed catalog for this route. */
|
||||
models?: readonly PiAiModelProfile[]
|
||||
/** Context capacity for a model neither the entry nor the catalog sizes. */
|
||||
defaultContextWindow: number
|
||||
/** Output capability for a model neither the entry nor the catalog sizes. */
|
||||
defaultMaxTokens: number
|
||||
}
|
||||
|
||||
/** Report a route the deployment cannot serve, naming the settings key at fault. */
|
||||
function invalid(provider: string, detail: string): never {
|
||||
throw new Error(`llm-pi-ai: provider "${provider}" ${detail}`)
|
||||
}
|
||||
|
||||
/**
|
||||
* The one wire protocol a catalog route's shipped models agree on. This is what
|
||||
* lets a deployment add a model the installed catalog has not caught up with —
|
||||
* a provider's newest release — without restating the protocol its siblings
|
||||
* already use. A route whose shipped models disagree (an OpenAI-style catalog
|
||||
* spanning Responses and Chat Completions) has no such answer, so a model it
|
||||
* does not describe must name its protocol at the route.
|
||||
*/
|
||||
function sharedCatalogApi(defaults: ReadonlyMap<string, Model<Api>>): string | undefined {
|
||||
const apis = new Set<string>()
|
||||
for (const model of defaults.values()) apis.add(model.api)
|
||||
return apis.size === 1 ? [...apis][0] : undefined
|
||||
}
|
||||
|
||||
/** One route's materialized catalog, plus the request caps its profile chose. */
|
||||
export interface RouteCatalog {
|
||||
/** The materialized models in configuration order. */
|
||||
models: readonly Model<Api>[]
|
||||
/**
|
||||
* Per-request output caps this profile explicitly configured, by model id.
|
||||
*
|
||||
* Separate from `Model.maxTokens` because the two answer different
|
||||
* questions: pi-ai requires `maxTokens` as the model's output *capability*,
|
||||
* while the harness seam's `defaultMaxTokens` is a cap the deployment chose
|
||||
* to send on requests that name none. Materializing a catalog capability as
|
||||
* a request default would start capping every request at a number nobody
|
||||
* picked, so only an explicit configuration lands here.
|
||||
*/
|
||||
configuredMaxTokens: ReadonlyMap<string, number>
|
||||
}
|
||||
|
||||
/**
|
||||
* Materialize one route's catalog by merging the installed catalog defaults
|
||||
* under the configured entries. A route with no configured `models` serves the
|
||||
* installed catalog unchanged, which is what keeps an existing
|
||||
* `providers: { deepseek: { apiKeyEnv: … } }` profile working untouched.
|
||||
* @param request - the route-level catalog facts.
|
||||
* @returns the materialized models and the explicitly configured request caps.
|
||||
*/
|
||||
export function resolveRouteModels(request: RouteCatalogRequest): RouteCatalog {
|
||||
const { provider } = request
|
||||
const defaults = catalogModels(provider)
|
||||
const providerBaseUrl = catalogProvider(provider)?.baseUrl
|
||||
// An absent `models` key and an empty one are the same request: the config
|
||||
// schema materializes `[]` for the absent case, and an empty catalog could
|
||||
// serve no request anyway, so both mean "serve the installed catalog".
|
||||
const configured = request.models ?? []
|
||||
const entries: readonly PiAiModelProfile[] = configured.length > 0
|
||||
? configured
|
||||
: [...defaults.values()].map(model => ({ id: model.id }))
|
||||
if (entries.length === 0) {
|
||||
invalid(provider, 'resolves no models; the installed catalog does not describe this route, so its models'
|
||||
+ ' must be listed in configuration')
|
||||
}
|
||||
const routeApi = sharedCatalogApi(defaults)
|
||||
const seen = new Set<string>()
|
||||
const configuredMaxTokens = new Map<string, number>()
|
||||
const models = entries.map((entry) => {
|
||||
if (entry.id.length === 0) invalid(provider, 'has a model with an empty id')
|
||||
if (seen.has(entry.id)) invalid(provider, `lists model "${entry.id}" more than once`)
|
||||
seen.add(entry.id)
|
||||
const base = defaults.get(entry.id)
|
||||
const api = request.api ?? base?.api ?? routeApi
|
||||
if (api === undefined) {
|
||||
invalid(provider, `model "${entry.id}" needs an api; the installed catalog does not describe it, so set the`
|
||||
+ ' route\'s api to the wire protocol its endpoint speaks')
|
||||
}
|
||||
const baseUrl = request.baseURL ?? base?.baseUrl ?? providerBaseUrl
|
||||
if (baseUrl === undefined) {
|
||||
invalid(provider, `model "${entry.id}" needs a baseURL; the installed catalog does not describe this route`)
|
||||
}
|
||||
// Capacities fall back to the route's own defaults, so a model listing that
|
||||
// discloses nothing but ids still yields a serviceable route. The fallback
|
||||
// is a guess by construction, which is why it is a configurable route field
|
||||
// rather than a constant buried here.
|
||||
const contextWindow = entry.contextWindow ?? base?.contextWindow ?? request.defaultContextWindow
|
||||
if (!Number.isInteger(contextWindow) || contextWindow <= 0) {
|
||||
invalid(provider, `model "${entry.id}" contextWindow must be a positive integer`)
|
||||
}
|
||||
const maxTokens = entry.maxTokens ?? base?.maxTokens ?? request.defaultMaxTokens
|
||||
if (!Number.isInteger(maxTokens) || maxTokens <= 0) {
|
||||
invalid(provider, `model "${entry.id}" maxTokens must be a positive integer`)
|
||||
}
|
||||
// Only a value the profile named is a deployment choice; the catalog's is
|
||||
// the model's capability and stays out of request defaults.
|
||||
if (entry.maxTokens !== undefined) configuredMaxTokens.set(entry.id, entry.maxTokens)
|
||||
return {
|
||||
// The installed entry lays the floor, and the fields below override it.
|
||||
// Enumerating instead would silently drop every `Model` field this
|
||||
// package does not model — reasoning-level spellings, compatibility
|
||||
// quirks, model headers, and whatever a pi-ai upgrade adds next. That is
|
||||
// not hypothetical: `headers` reached this file only after an nvidia
|
||||
// route lost it, and a rebuild keeps re-earning that bug on every
|
||||
// upgrade.
|
||||
...base,
|
||||
id: entry.id,
|
||||
name: entry.name ?? base?.name ?? entry.id,
|
||||
api,
|
||||
provider,
|
||||
baseUrl,
|
||||
// Reasoning rides the installed entry or is absent: a bare boolean would
|
||||
// make pi-ai advertise effort levels with no `thinkingLevelMap` to spell
|
||||
// them, and no listing endpoint reports a model's reasoning protocol.
|
||||
reasoning: base?.reasoning ?? false,
|
||||
input: base?.input ?? TEXT_ONLY,
|
||||
cost: base?.cost ?? NO_COST,
|
||||
contextWindow,
|
||||
maxTokens,
|
||||
}
|
||||
})
|
||||
return { models, configuredMaxTokens }
|
||||
}
|
||||
@@ -3,29 +3,77 @@
|
||||
* Profiles are a dict keyed by provider route, so the composition base and a
|
||||
* user-settings layer merge per provider and the route set is structural.
|
||||
*
|
||||
* A route key is not required to name an installed pi-ai provider. When it does,
|
||||
* that provider's endpoint, protocol, display name, and model catalog are the
|
||||
* profile's defaults and the profile overrides them field by field; when it does
|
||||
* not, the profile is the whole provider declaration. Resolution therefore ends
|
||||
* in a built pi-ai `Provider` per route: everything a request needs is decided
|
||||
* once, while the configuration key that made a route unserviceable can still be
|
||||
* named in the failure.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/config
|
||||
*/
|
||||
|
||||
import { getBuiltinProviders } from '@earendil-works/pi-ai/providers/all'
|
||||
import type { CacheRetention, ModelThinkingLevel, ThinkingBudgets, Transport } from '@earendil-works/pi-ai'
|
||||
import type { CacheRetention, ModelThinkingLevel, Provider, ThinkingBudgets, Transport } from '@earendil-works/pi-ai'
|
||||
import z from 'schemastery'
|
||||
import { credentialRef } from '@deepseek-ai/dsh-credentials'
|
||||
import type { CredentialRef } from '@deepseek-ai/dsh-credentials'
|
||||
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
|
||||
import { resolveRetryPolicy, RetryPolicySchema } from '@deepseek-ai/dsh-llm'
|
||||
import { normalizeApiKey, resolveRetryPolicy, RetryPolicySchema } from '@deepseek-ai/dsh-llm'
|
||||
import type { ResolvedRetryPolicy, RetryPolicyConfig } from '@deepseek-ai/dsh-llm'
|
||||
import { resolveRouteModels } from './catalog.ts'
|
||||
import type { PiAiModelProfile } from './catalog.ts'
|
||||
import { buildProvider, supportedProtocols } from './provider.ts'
|
||||
|
||||
/** Default maximum idle interval while an adapter stream read is outstanding. */
|
||||
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000
|
||||
|
||||
/** Context capacity assumed for a model neither configuration nor the catalog sizes. */
|
||||
export const DEFAULT_CONTEXT_WINDOW = 262_144
|
||||
|
||||
/** Output capability assumed for a model neither configuration nor the catalog sizes. */
|
||||
export const DEFAULT_MAX_TOKENS = 32_768
|
||||
|
||||
export type { PiAiModelProfile } from './catalog.ts'
|
||||
|
||||
/** Configuration for one pi-ai provider route; the `providers` dict key IS the route. */
|
||||
export interface PiAiProviderProfile {
|
||||
/** Literal provider credential; prefer {@link apiKeyEnv}. With both absent pi-ai uses its provider-native ambient discovery. */
|
||||
/**
|
||||
* Literal provider credential; prefer {@link apiKeyEnv}. With both absent pi-ai uses its
|
||||
* provider-native ambient discovery. Trimmed and format-checked by {@link resolveProfiles}; a
|
||||
* value no HTTP header can carry fails there rather than inside `fetch`.
|
||||
*/
|
||||
apiKey?: string
|
||||
/** Credential reference (environment-variable name) resolved per request through `ctx.credentials`. */
|
||||
apiKeyEnv?: string
|
||||
/** Override the selected catalog model's endpoint without changing its protocol metadata. */
|
||||
/** Name shown by configuration surfaces; defaults to the route key. */
|
||||
displayName?: string
|
||||
/**
|
||||
* Wire protocol every model on this route speaks. Omission keeps each
|
||||
* installed catalog model's own protocol, which is why a catalog route needs
|
||||
* no protocol at all; a route the catalog does not ship must name one.
|
||||
*/
|
||||
api?: string
|
||||
/** Endpoint for this route's models; defaults to the installed catalog's endpoint. */
|
||||
baseURL?: string
|
||||
/**
|
||||
* This route's model catalog. Omission serves the installed catalog for the
|
||||
* route unchanged; an explicit list replaces it, each entry defaulting its
|
||||
* unset fields from the installed model of the same id.
|
||||
*/
|
||||
models?: PiAiModelProfile[]
|
||||
/**
|
||||
* Context capacity for a model this route lists that neither the entry nor
|
||||
* the installed catalog sizes (default 262,144). A guess by construction, so
|
||||
* a deployment whose gateway serves smaller models corrects it here.
|
||||
*/
|
||||
defaultContextWindow?: number
|
||||
/**
|
||||
* Output capability for a model this route lists that neither the entry nor
|
||||
* the installed catalog sizes (default 32,768). This sizes the model; it
|
||||
* never becomes a per-request cap on its own.
|
||||
*/
|
||||
defaultMaxTokens?: number
|
||||
/** Provider request headers; Harness attribution wins reserved names. */
|
||||
headers?: Record<string, string>
|
||||
/** Provider-neutral pi-ai reasoning level. */
|
||||
@@ -47,15 +95,31 @@ export interface PiAiProviderProfile {
|
||||
}
|
||||
|
||||
/** Validated profile with its route stamped and every adapter-owned default resolved. */
|
||||
export interface ResolvedPiAiProviderProfile extends Omit<PiAiProviderProfile, 'apiKeyEnv' | 'retryPolicy'> {
|
||||
/** pi-ai provider catalog name and Harness route key (the configuration dict key). */
|
||||
export interface ResolvedPiAiProviderProfile
|
||||
extends Omit<PiAiProviderProfile, 'apiKeyEnv' | 'retryPolicy' | 'models' | 'displayName'> {
|
||||
/** Harness route key and the `Models` collection key (the configuration dict key). */
|
||||
provider: string
|
||||
/** Resolved display name for selectors and configuration surfaces. */
|
||||
displayName: string
|
||||
/** Validated credential reference, when one is configured. */
|
||||
apiKeyEnv?: CredentialRef
|
||||
/** Positive finite provider-idle interval after defaulting. */
|
||||
streamIdleTimeoutMs: number
|
||||
/** Immutable retry policy captured with this provider route. */
|
||||
retryPolicy: ResolvedRetryPolicy
|
||||
/**
|
||||
* The pi-ai provider this route registers, built from the resolved models.
|
||||
* Construction happens here so an unserviceable protocol or an underspecified
|
||||
* model fails with the rest of resolution, leaving the last good route set
|
||||
* serving requests.
|
||||
*/
|
||||
piProvider: Provider
|
||||
/**
|
||||
* Per-request output caps this profile explicitly configured, by model id.
|
||||
* The seam materializes one only into a request that names no cap of its
|
||||
* own, so a catalog capability must not appear here.
|
||||
*/
|
||||
configuredMaxTokens: ReadonlyMap<string, number>
|
||||
}
|
||||
|
||||
/** Plugin configuration: the provider routes this instance owns. */
|
||||
@@ -75,10 +139,22 @@ const thinkingBudgets = z.object({
|
||||
high: z.number(),
|
||||
})
|
||||
|
||||
const modelProfile: z<PiAiModelProfile> = z.object({
|
||||
id: z.string().required(),
|
||||
name: z.string(),
|
||||
contextWindow: z.number().step(1).min(1),
|
||||
maxTokens: z.number().step(1).min(1),
|
||||
})
|
||||
|
||||
const profile = z.object({
|
||||
apiKey: z.string().role('secret'),
|
||||
apiKeyEnv: z.string().role('credential-ref'),
|
||||
displayName: z.string(),
|
||||
api: z.union(supportedProtocols()),
|
||||
baseURL: z.string(),
|
||||
models: z.array(modelProfile),
|
||||
defaultContextWindow: z.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
|
||||
defaultMaxTokens: z.number().step(1).min(1).default(DEFAULT_MAX_TOKENS),
|
||||
headers: z.dict(z.string()),
|
||||
reasoning: z.union(['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max']),
|
||||
thinkingBudgets,
|
||||
@@ -96,10 +172,44 @@ export const Config: z<Config> = z.object({
|
||||
})
|
||||
|
||||
/**
|
||||
* Validate profiles against the installed pi-ai catalog and return a detached
|
||||
* route-keyed map suitable for per-request reads. This is the one explicit
|
||||
* resolve step, so an omitted dict resolves to the empty (dormant) route set
|
||||
* here rather than through a hidden fallback.
|
||||
* Reject a section this adapter could not serve. Registered as the settings
|
||||
* namespace's validator, so an unserviceable profile is refused where it is
|
||||
* *written* — `settings.mutate` answers `settings-rejected` with the offending
|
||||
* route and model named — instead of being stored and then quietly disabling
|
||||
* every route in the namespace. It stays a validator rather than a schema
|
||||
* transform because the schema is also the shape a configuration surface
|
||||
* renders and the value an absent section resolves to; wrapping it would break
|
||||
* both.
|
||||
* @param config - the resolved section to check.
|
||||
* @throws Error naming the route and model that cannot be served.
|
||||
*/
|
||||
export function assertServiceable(config: Config): void {
|
||||
resolveProfiles(config.providers)
|
||||
}
|
||||
|
||||
/** Reject a pre-release profile shape, naming the replacement. */
|
||||
function rejectRemovedFields(provider: string, source: PiAiProviderProfile): void {
|
||||
const legacy = source as PiAiProviderProfile & {
|
||||
provider?: unknown
|
||||
maxRetries?: unknown
|
||||
maxRetryDelayMs?: unknown
|
||||
}
|
||||
if ('provider' in legacy) {
|
||||
throw new Error(`llm-pi-ai: provider "${provider}" sets "provider", which moved to the providers dict key`)
|
||||
}
|
||||
if ('maxRetries' in legacy || 'maxRetryDelayMs' in legacy) {
|
||||
throw new Error(
|
||||
`llm-pi-ai: provider "${provider}" sets maxRetries or maxRetryDelayMs, which were removed;`
|
||||
+ ' compose agent recovery with dsh-llm-retry',
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate profiles and return a detached route-keyed map suitable for
|
||||
* per-request reads. This is the one explicit resolve step, so an omitted dict
|
||||
* resolves to the empty (dormant) route set here rather than through a hidden
|
||||
* fallback, and each route's models and pi-ai provider are materialized once.
|
||||
* @param providers - configured provider profiles keyed by route.
|
||||
* @returns validated profiles in configuration order.
|
||||
*/
|
||||
@@ -110,28 +220,29 @@ export function resolveProfiles(
|
||||
throw new Error('llm-pi-ai: providers is now a dict keyed by provider route, not an array of profiles')
|
||||
}
|
||||
const entries = Object.entries(providers ?? {})
|
||||
const supported = new Set<string>(getBuiltinProviders())
|
||||
const resolved = new Map<string, ResolvedPiAiProviderProfile>()
|
||||
for (const [provider, source] of entries) {
|
||||
const legacy = source as PiAiProviderProfile & {
|
||||
provider?: unknown
|
||||
maxRetries?: unknown
|
||||
maxRetryDelayMs?: unknown
|
||||
}
|
||||
if ('provider' in legacy) {
|
||||
throw new Error('llm-pi-ai: the profile "provider" field moved to the providers dict key')
|
||||
}
|
||||
if ('maxRetries' in legacy || 'maxRetryDelayMs' in legacy) {
|
||||
throw new Error('llm-pi-ai: maxRetries and maxRetryDelayMs were removed; compose agent recovery with dsh-llm-retry')
|
||||
}
|
||||
rejectRemovedFields(provider, source)
|
||||
if (provider.length === 0) throw new Error('llm-pi-ai: provider names must be non-empty')
|
||||
if (!supported.has(provider)) throw new Error(`llm-pi-ai: unknown pi-ai provider "${provider}"`)
|
||||
if (source.apiKey !== undefined && source.apiKey.trim().length === 0) {
|
||||
throw new Error(`llm-pi-ai: provider "${provider}" has an empty apiKey; omit it to use ambient authentication`)
|
||||
// Omission selects the installed provider's own auth — ambient discovery
|
||||
// or OAuth — so only a supplied key is judged.
|
||||
let apiKey: string | undefined
|
||||
if (source.apiKey !== undefined) {
|
||||
const checked = normalizeApiKey(source.apiKey)
|
||||
if (!checked.ok) {
|
||||
throw new Error(checked.reason === 'empty'
|
||||
? `llm-pi-ai: provider "${provider}" has an empty apiKey; omit it to use ambient authentication`
|
||||
: `llm-pi-ai: provider "${provider}" has an apiKey containing characters no HTTP header can carry;`
|
||||
+ ' paste the raw key only')
|
||||
}
|
||||
apiKey = checked.value
|
||||
}
|
||||
if (source.baseURL !== undefined && source.baseURL.length === 0) {
|
||||
throw new Error(`llm-pi-ai: provider "${provider}" has an empty baseURL`)
|
||||
}
|
||||
if (source.displayName !== undefined && source.displayName.length === 0) {
|
||||
throw new Error(`llm-pi-ai: provider "${provider}" has an empty displayName`)
|
||||
}
|
||||
const streamIdleTimeoutMs = source.streamIdleTimeoutMs ?? DEFAULT_STREAM_IDLE_TIMEOUT_MS
|
||||
if (!Number.isFinite(streamIdleTimeoutMs)
|
||||
|| streamIdleTimeoutMs <= 0
|
||||
@@ -140,15 +251,38 @@ export function resolveProfiles(
|
||||
`llm-pi-ai: provider "${provider}" streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`,
|
||||
)
|
||||
}
|
||||
const { apiKeyEnv, retryPolicy, ...rest } = source
|
||||
// The route key, not the installed provider's own name: the directory has
|
||||
// always shown route keys, and a catalog route must not silently rename
|
||||
// itself on every configuration surface just because it gained a profile.
|
||||
const displayName = source.displayName ?? provider
|
||||
const catalog = resolveRouteModels({
|
||||
provider,
|
||||
...source.api === undefined ? {} : { api: source.api },
|
||||
...source.baseURL === undefined ? {} : { baseURL: source.baseURL },
|
||||
...source.models === undefined ? {} : { models: source.models },
|
||||
defaultContextWindow: source.defaultContextWindow ?? DEFAULT_CONTEXT_WINDOW,
|
||||
defaultMaxTokens: source.defaultMaxTokens ?? DEFAULT_MAX_TOKENS,
|
||||
})
|
||||
const { apiKeyEnv, retryPolicy, models: _models, displayName: _displayName, ...rest } = source
|
||||
resolved.set(provider, {
|
||||
...rest,
|
||||
...apiKey === undefined ? {} : { apiKey },
|
||||
provider,
|
||||
displayName,
|
||||
...apiKeyEnv === undefined ? {} : { apiKeyEnv: credentialRef(apiKeyEnv) },
|
||||
streamIdleTimeoutMs,
|
||||
retryPolicy: resolveRetryPolicy(retryPolicy, `llm-pi-ai: provider "${provider}" retryPolicy`),
|
||||
...rest.headers === undefined ? {} : { headers: { ...rest.headers } },
|
||||
...rest.thinkingBudgets === undefined ? {} : { thinkingBudgets: { ...rest.thinkingBudgets } },
|
||||
configuredMaxTokens: catalog.configuredMaxTokens,
|
||||
piProvider: buildProvider({
|
||||
provider,
|
||||
displayName,
|
||||
...source.api === undefined ? {} : { api: source.api },
|
||||
...source.baseURL === undefined ? {} : { baseURL: source.baseURL },
|
||||
models: catalog.models,
|
||||
namesCredential: source.apiKey !== undefined || apiKeyEnv !== undefined,
|
||||
}),
|
||||
})
|
||||
}
|
||||
return resolved
|
||||
|
||||
284
packages/llm/llm-pi-ai/src/discovery.ts
Normal file
284
packages/llm/llm-pi-ai/src/discovery.ts
Normal file
@@ -0,0 +1,284 @@
|
||||
/**
|
||||
* Answering "which models can this provider serve?" for the configuration
|
||||
* surface's "fetch available models" action.
|
||||
*
|
||||
* A route the installed pi-ai catalog ships is answered **from that catalog**,
|
||||
* with no network call at all: pi-ai's registry is the authoritative list for
|
||||
* its own providers, and it carries the capacities a listing endpoint would
|
||||
* not disclose. Only a route the catalog does not describe — a gateway, a
|
||||
* self-hosted server — is interrogated over the wire.
|
||||
*
|
||||
* Neither path is a catalog refresh. Nothing here is stored: the request
|
||||
* carries a draft the user is still editing, and the reply is candidate
|
||||
* metadata the surface offers for adoption. `settings.yaml` remains the only
|
||||
* thing that decides what a route serves.
|
||||
*
|
||||
* Only OpenAI-compatible protocols are interrogated. Their listing is the one
|
||||
* shape a gateway, a self-hosted server, and the official endpoints all agree
|
||||
* on, which is the case this action exists for; every other protocol reports
|
||||
* that it cannot be interrogated so the surface falls back to hand-entry
|
||||
* rather than guessing a response shape.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/discovery
|
||||
*/
|
||||
|
||||
import { INVALID_CREDENTIAL_CODE, LlmError, normalizeApiKey } from '@deepseek-ai/dsh-llm'
|
||||
import type { LlmDiscoveredModel, LlmModelDiscoveryRequest } from '@deepseek-ai/dsh-llm'
|
||||
import { attributionHeaders } from '@deepseek-ai/dsh-llm'
|
||||
import { catalogModels } from './catalog.ts'
|
||||
|
||||
/**
|
||||
* Protocols whose model listing this module can read: the two that speak
|
||||
* OpenAI's `GET /models` shape with bearer auth. Azure is absent despite its
|
||||
* OpenAI lineage — it authenticates with an `api-key` header and requires an
|
||||
* `api-version` query — and Codex authenticates through OAuth; guessing at
|
||||
* either would report an authentication failure as a provider with no models.
|
||||
* pi-ai's remaining protocols are absent for the same reason.
|
||||
*/
|
||||
const LISTABLE_PROTOCOLS: ReadonlySet<string> = new Set([
|
||||
'openai-completions',
|
||||
'openai-responses',
|
||||
])
|
||||
|
||||
/**
|
||||
* Endpoint replies larger than this are refused. The endpoint is whatever URL
|
||||
* the user typed, so the ceiling holds on the bytes actually read rather than
|
||||
* on the length the server claims — the same two-stage shape `dsh-web-fetch`
|
||||
* uses for its own caller-supplied URLs, except that a truncated model listing
|
||||
* is not parseable, so overflow rejects instead of truncating.
|
||||
*/
|
||||
const MAX_RESPONSE_BYTES = 4 * 1024 * 1024
|
||||
|
||||
/** One entry of an OpenAI-compatible `GET /models` reply. */
|
||||
interface ListingEntry {
|
||||
id?: unknown
|
||||
/** Common gateway extensions; absent from the official listings. */
|
||||
name?: unknown
|
||||
display_name?: unknown
|
||||
context_window?: unknown
|
||||
context_length?: unknown
|
||||
max_tokens?: unknown
|
||||
max_output_tokens?: unknown
|
||||
}
|
||||
|
||||
/** A positive integer field of a listing entry, or `undefined` when absent or unusable. */
|
||||
function capacity(...candidates: readonly unknown[]): number | undefined {
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === 'number' && Number.isInteger(candidate) && candidate > 0) return candidate
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/** A non-empty string field of a listing entry, or `undefined`. */
|
||||
function label(...candidates: readonly unknown[]): string | undefined {
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === 'string' && candidate.length > 0) return candidate
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Join the endpoint base with the listing path. The base is treated as a
|
||||
* prefix rather than a URL to resolve against, so a deployment path such as
|
||||
* `https://gateway.example/openai/v1` keeps its segments instead of losing
|
||||
* them to `URL` resolution.
|
||||
*/
|
||||
function listingUrl(baseURL: string): string {
|
||||
return `${baseURL.replace(/\/+$/, '')}/models`
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a reply body, refusing one that outgrows the ceiling. A declared length
|
||||
* is checked first so an honest server is turned away without transferring
|
||||
* anything; the accumulated total is what actually enforces the bound, because
|
||||
* a server that under-declares (or streams) tells us nothing up front.
|
||||
*/
|
||||
async function readBounded(response: Response, url: string): Promise<string> {
|
||||
const oversized = (): LlmError =>
|
||||
new LlmError(`${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, 'DISCOVERY_FAILED')
|
||||
const declared = Number(response.headers.get('content-length') ?? Number.NaN)
|
||||
if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
|
||||
await response.body?.cancel()
|
||||
throw oversized()
|
||||
}
|
||||
/* v8 ignore next -- fetch always exposes a body stream on a 2xx Response; the null guard is defensive. */
|
||||
if (response.body === null) return ''
|
||||
const reader = response.body.getReader()
|
||||
const chunks: Uint8Array[] = []
|
||||
let total = 0
|
||||
try {
|
||||
for (;;) {
|
||||
const { done, value } = await reader.read()
|
||||
if (done) break
|
||||
total += value.byteLength
|
||||
if (total > MAX_RESPONSE_BYTES) throw oversized()
|
||||
chunks.push(value)
|
||||
}
|
||||
} finally {
|
||||
/* v8 ignore next 4 -- cancel() after a completed or abandoned read settles without rejecting; unobserved best-effort cleanup. */
|
||||
await reader.cancel().catch(() => {
|
||||
// Cancel after a drained read, or after this function walked away from
|
||||
// an oversized one, is cleanup; the reply is already decided either way.
|
||||
})
|
||||
}
|
||||
const body = new Uint8Array(total)
|
||||
let offset = 0
|
||||
for (const chunk of chunks) {
|
||||
body.set(chunk, offset)
|
||||
offset += chunk.byteLength
|
||||
}
|
||||
return new TextDecoder().decode(body)
|
||||
}
|
||||
|
||||
/**
|
||||
* Read one OpenAI-compatible listing reply. Entries without a usable id are
|
||||
* skipped rather than failing the whole interrogation: a single malformed row
|
||||
* should not deny the user the rest of a working endpoint's catalog.
|
||||
*/
|
||||
function readListing(body: unknown): LlmDiscoveredModel[] {
|
||||
const data = (body as { data?: unknown } | null)?.data
|
||||
if (!Array.isArray(data)) {
|
||||
throw new LlmError(
|
||||
'the endpoint\'s model listing has no "data" array; enter this provider\'s models by hand',
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
const models: LlmDiscoveredModel[] = []
|
||||
for (const raw of data) {
|
||||
const entry = raw as ListingEntry | null
|
||||
const id = label(entry?.id)
|
||||
if (id === undefined) continue
|
||||
const name = label(entry?.name, entry?.display_name)
|
||||
const contextWindow = capacity(entry?.context_window, entry?.context_length)
|
||||
const maxTokens = capacity(entry?.max_output_tokens, entry?.max_tokens)
|
||||
models.push({
|
||||
id,
|
||||
...name === undefined ? {} : { name },
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
...maxTokens === undefined ? {} : { maxTokens },
|
||||
})
|
||||
}
|
||||
return models
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept one probe key, or refuse it before the header is built. Without this
|
||||
* the `fetch` below would throw a ByteString `TypeError` that this function's
|
||||
* catch reports as `could not reach <url>` — blaming the network for a local,
|
||||
* deterministic fault.
|
||||
* @param raw - the key typed into the form or read from storage.
|
||||
* @returns the trimmed, usable key.
|
||||
*/
|
||||
function usableProbeKey(raw: string): string {
|
||||
const checked = normalizeApiKey(raw)
|
||||
if (checked.ok) return checked.value
|
||||
throw new LlmError(
|
||||
checked.reason === 'empty'
|
||||
? 'this provider\'s API key is blank; enter it on the Models page, or clear it to probe unauthenticated'
|
||||
: 'this provider\'s API key contains characters no HTTP header can carry; paste the raw key only',
|
||||
INVALID_CREDENTIAL_CODE,
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Interrogate one draft provider endpoint for the models it advertises.
|
||||
* @param request - the endpoint, protocol, and one-shot credential to use.
|
||||
* @param storedApiKey - the credential the named route already stored, asked
|
||||
* for only when the draft carries none and only on the path that reaches the
|
||||
* network. A configuration surface never holds a stored secret — it edits a
|
||||
* redacted descriptor — so without this an already-configured route would be
|
||||
* interrogated unauthenticated and answer 401.
|
||||
* @returns the advertised models in endpoint order.
|
||||
* @throws LlmError when the protocol has no readable listing, the endpoint
|
||||
* refuses or fails the request, or the reply is not a model listing.
|
||||
*/
|
||||
export async function discoverModels(
|
||||
request: LlmModelDiscoveryRequest,
|
||||
storedApiKey?: () => Promise<string | undefined>,
|
||||
): Promise<readonly LlmDiscoveredModel[]> {
|
||||
// A catalog route already has its answer, and a better one: the installed
|
||||
// entries carry context windows and output caps no listing endpoint reports.
|
||||
if (request.provider !== undefined) {
|
||||
const installed = catalogModels(request.provider)
|
||||
if (installed.size > 0) {
|
||||
return [...installed.values()].map(model => ({
|
||||
id: model.id,
|
||||
name: model.name,
|
||||
contextWindow: model.contextWindow,
|
||||
maxTokens: model.maxTokens,
|
||||
}))
|
||||
}
|
||||
}
|
||||
if (request.baseURL === undefined || request.baseURL.length === 0) {
|
||||
throw new LlmError(
|
||||
`pi-ai ships no catalog for provider "${request.provider ?? ''}", so its models can only come from its`
|
||||
+ " endpoint; set a baseURL, or enter this provider's models by hand",
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
// A draft that has not chosen a protocol yet is asked as OpenAI Chat
|
||||
// Completions: it is the shape a gateway is overwhelmingly likely to speak,
|
||||
// and the alternative — refusing until the field is filled — would withhold
|
||||
// the action from the case it exists for. The cost is a misdirected message
|
||||
// when the endpoint speaks something else (an Anthropic gateway answers 401,
|
||||
// which reads as a credential problem), and hand-entry remains the way out.
|
||||
const api = request.api ?? 'openai-completions'
|
||||
if (!LISTABLE_PROTOCOLS.has(api)) {
|
||||
throw new LlmError(
|
||||
`pi-ai protocol "${api}" has no model listing this build can read; enter this provider's models by hand`,
|
||||
'DISCOVERY_UNSUPPORTED',
|
||||
)
|
||||
}
|
||||
const url = listingUrl(request.baseURL)
|
||||
// A key typed into the form wins: it is the one the user is testing, and it
|
||||
// may be the replacement for exactly the stored key that is failing. The
|
||||
// stored one is only asked for here, past the catalog short-circuit and the
|
||||
// protocol check, so a route answered from the registry costs no credential
|
||||
// lookup — and no diagnostic about a credential it never needed.
|
||||
// A probe carrying no key stays unauthenticated, which is how a route that
|
||||
// relies on the provider's own ambient discovery is meant to be asked.
|
||||
const supplied = request.apiKey ?? await storedApiKey?.()
|
||||
const apiKey = supplied === undefined ? undefined : usableProbeKey(supplied)
|
||||
let response: Response
|
||||
try {
|
||||
response = await fetch(url, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
accept: 'application/json',
|
||||
...apiKey === undefined ? {} : { authorization: `Bearer ${apiKey}` },
|
||||
...attributionHeaders(),
|
||||
},
|
||||
...request.signal === undefined ? {} : { signal: request.signal },
|
||||
})
|
||||
} catch (error: unknown) {
|
||||
if (request.signal?.aborted) {
|
||||
throw new LlmError('model discovery aborted by caller', 'ABORTED', { cause: error })
|
||||
}
|
||||
throw new LlmError(`could not reach ${url}`, 'DISCOVERY_FAILED', { cause: error })
|
||||
}
|
||||
if (!response.ok) {
|
||||
throw new LlmError(
|
||||
`${url} answered ${response.status}${response.status === 401 || response.status === 403 ? '; check the API key' : ''}`,
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
let text: string
|
||||
try {
|
||||
text = await readBounded(response, url)
|
||||
} catch (error: unknown) {
|
||||
// Cancellation during the body read rejects with the abort reason, which
|
||||
// may be any value; the caller gets the same coded failure it would have
|
||||
// for a cancellation before the request went out.
|
||||
if (request.signal?.aborted) {
|
||||
throw new LlmError('model discovery aborted by caller', 'ABORTED', { cause: error })
|
||||
}
|
||||
throw error
|
||||
}
|
||||
let body: unknown
|
||||
try {
|
||||
body = JSON.parse(text)
|
||||
} catch (error: unknown) {
|
||||
throw new LlmError(`${url} did not answer with JSON`, 'DISCOVERY_FAILED', { cause: error })
|
||||
}
|
||||
return readListing(body)
|
||||
}
|
||||
@@ -1,10 +1,11 @@
|
||||
/**
|
||||
* Generic pi-ai-backed LLM adapter plugin. One plugin instance owns a dict of
|
||||
* provider routes; requests select a profile by provider and resolve the
|
||||
* model dynamically from pi-ai's installed catalog. Profile facts resolve per
|
||||
* request over the optional `llm-pi-ai` user-settings section and the
|
||||
* optional credential seam, so a changed key, endpoint, or knob reaches the
|
||||
* next request without a restart; a changed *route set* (or a route's
|
||||
* provider routes; a route naming an installed pi-ai provider inherits that
|
||||
* provider's endpoint, protocol, and model catalog as defaults, and a route
|
||||
* pi-ai does not ship is declared outright. Profile facts resolve per request
|
||||
* over the optional `llm-pi-ai` user-settings section and the optional
|
||||
* credential seam, so a changed key, endpoint, model, or knob reaches the next
|
||||
* request without a restart; a changed *route set* (or a route's
|
||||
* registration-captured retry policy) re-registers the same adapter instance
|
||||
* in place.
|
||||
*
|
||||
@@ -13,34 +14,49 @@
|
||||
* name: '@deepseek-ai/dsh-llm-pi-ai'
|
||||
* config:
|
||||
* providers:
|
||||
* # Catalog route: everything but the credential comes from pi-ai.
|
||||
* openai:
|
||||
* apiKeyEnv: OPENAI_API_KEY
|
||||
* retryPolicy:
|
||||
* mode: normal
|
||||
* maxRetries: 2
|
||||
* # Catalog route with the catalog narrowed and one capacity corrected.
|
||||
* anthropic:
|
||||
* apiKeyEnv: ANTHROPIC_API_KEY
|
||||
* openrouter:
|
||||
* apiKeyEnv: OPENROUTER_API_KEY
|
||||
* baseURL: https://proxy.example.com/v1
|
||||
* models:
|
||||
* - id: claude-sonnet-4-5
|
||||
* contextWindow: 200000
|
||||
* # Hand-declared route: pi-ai ships nothing under this key.
|
||||
* acme-gateway:
|
||||
* displayName: Acme Gateway
|
||||
* apiKeyEnv: ACME_GATEWAY_API_KEY
|
||||
* api: openai-completions
|
||||
* baseURL: https://gateway.acme.example/v1
|
||||
* models:
|
||||
* - id: acme-large
|
||||
* name: Acme Large
|
||||
* contextWindow: 65536
|
||||
* maxTokens: 4096
|
||||
* ```
|
||||
*
|
||||
* @module @deepseek-ai/dsh-llm-pi-ai
|
||||
*/
|
||||
|
||||
import type { Context } from 'cordis'
|
||||
import { getBuiltinProviders } from '@earendil-works/pi-ai/providers/all'
|
||||
import { LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import type { AdapterRegistrationHandle } from '@deepseek-ai/dsh-llm'
|
||||
import { assertUsableApiKey, LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import type { AdapterRegistrationHandle, DirectoryRegistrationHandle, LlmConfigurableProvider } from '@deepseek-ai/dsh-llm'
|
||||
import { deepEqualJson, installSettingsSection, settingsNamespace } from '@deepseek-ai/dsh-settings'
|
||||
import { PiAiAdapter } from './adapter.ts'
|
||||
import { Config, resolveProfiles } from './config.ts'
|
||||
import { catalogProviderIds } from './catalog.ts'
|
||||
import { assertServiceable, Config, resolveProfiles } from './config.ts'
|
||||
import type { ResolvedPiAiProviderProfile } from './config.ts'
|
||||
import { discoverModels } from './discovery.ts'
|
||||
|
||||
export { PiAiAdapter } from './adapter.ts'
|
||||
export type { PiAiAdapterOptions } from './adapter.ts'
|
||||
export { Config } from './config.ts'
|
||||
export type { PiAiProviderProfile, ResolvedPiAiProviderProfile } from './config.ts'
|
||||
export type { PiAiModelProfile, PiAiProviderProfile, ResolvedPiAiProviderProfile } from './config.ts'
|
||||
export { supportedProtocols } from './provider.ts'
|
||||
|
||||
export const name = 'llm-pi-ai'
|
||||
export const inject = ['llm']
|
||||
@@ -54,33 +70,70 @@ const NS = settingsNamespace('llm-pi-ai')
|
||||
*/
|
||||
function registrationFacts(profiles: ReadonlyMap<string, ResolvedPiAiProviderProfile>): unknown {
|
||||
return [...profiles.entries()]
|
||||
.map(([provider, profile]) => ({ provider, retryPolicy: profile.retryPolicy }))
|
||||
// `displayName` rides along because the registry hands it to every selector
|
||||
// through `providerInfo()`: a rename that did not re-register would leave
|
||||
// the old label showing until some unrelated fact happened to change.
|
||||
.map(([provider, profile]) => ({
|
||||
provider,
|
||||
displayName: profile.displayName,
|
||||
retryPolicy: profile.retryPolicy,
|
||||
}))
|
||||
.sort((left, right) => left.provider.localeCompare(right.provider))
|
||||
}
|
||||
|
||||
/**
|
||||
* The configurable-provider directory: every installed catalog route, plus
|
||||
* every route the current profiles declare. A hand-declared route has no
|
||||
* catalog entry, so without this union it would have no settings address and
|
||||
* configuration surfaces could neither show nor edit it.
|
||||
* @param profiles - the currently resolved provider profiles.
|
||||
* @returns the directory entries in catalog order, declared routes last.
|
||||
*/
|
||||
function directoryEntries(
|
||||
profiles: ReadonlyMap<string, ResolvedPiAiProviderProfile>,
|
||||
): LlmConfigurableProvider[] {
|
||||
const catalog = new Set(catalogProviderIds())
|
||||
const entries = new Map<string, LlmConfigurableProvider>()
|
||||
const declare = (provider: string, displayName: string): void => {
|
||||
entries.set(provider, {
|
||||
provider,
|
||||
displayName,
|
||||
settingsNs: NS,
|
||||
settingsPath: ['providers', provider],
|
||||
// Membership of the installed catalog, not of the settings document:
|
||||
// narrowing a shipped provider's models stores a profile too, and that
|
||||
// route is still one pi-ai knows.
|
||||
declared: !catalog.has(provider),
|
||||
})
|
||||
}
|
||||
for (const provider of catalog) declare(provider, provider)
|
||||
for (const [provider, profile] of profiles) declare(provider, profile.displayName)
|
||||
return [...entries.values()]
|
||||
}
|
||||
|
||||
/** Register one generic pi-ai adapter for all configured provider routes. */
|
||||
export function apply(ctx: Context, config: Config): void {
|
||||
let current: () => Config = () => config
|
||||
let lastRaw: Config | undefined
|
||||
let lastGood: ReadonlyMap<string, ResolvedPiAiProviderProfile> | undefined
|
||||
let memoized: ReadonlyMap<string, ResolvedPiAiProviderProfile> | undefined
|
||||
/**
|
||||
* The resolved profiles for the current configuration, memoized by the raw
|
||||
* snapshot's identity — which is also what makes the adapter's own snapshot
|
||||
* stable across operations that observe no change.
|
||||
*
|
||||
* No fallback for an unserviceable snapshot lives here: the section schema
|
||||
* resolves the whole profile set, so a write that could not be served is
|
||||
* refused where it is written, and the settings seam keeps a namespace's
|
||||
* last good value for a stored section that fails. Anything reaching this
|
||||
* point has already resolved once.
|
||||
*/
|
||||
const profiles = (): ReadonlyMap<string, ResolvedPiAiProviderProfile> => {
|
||||
const raw = current()
|
||||
if (raw === lastRaw && lastGood !== undefined) return lastGood
|
||||
try {
|
||||
const next = resolveProfiles(raw.providers)
|
||||
lastRaw = raw
|
||||
lastGood = next
|
||||
return next
|
||||
} catch (error) {
|
||||
// Static composition resolves before anything registers, so this branch
|
||||
// only sees a live settings snapshot failing catalog or bound checks:
|
||||
// keep serving the last good profiles and say so once per bad snapshot.
|
||||
if (lastGood === undefined) throw error
|
||||
lastRaw = raw
|
||||
ctx.logger.error('llm-pi-ai: keeping the last good profiles after an invalid settings section')
|
||||
ctx.logger.error(error)
|
||||
return lastGood
|
||||
}
|
||||
if (raw === lastRaw && memoized !== undefined) return memoized
|
||||
const next = resolveProfiles(raw.providers)
|
||||
lastRaw = raw
|
||||
memoized = next
|
||||
return next
|
||||
}
|
||||
profiles()
|
||||
|
||||
@@ -102,7 +155,7 @@ export function apply(ctx: Context, config: Config): void {
|
||||
// Without the seam, read exactly the named variable so a plain
|
||||
// cordis.yml composition works from the environment alone.
|
||||
: process.env[ref]
|
||||
if (hit !== undefined && hit.length > 0) return hit
|
||||
if (hit !== undefined && hit.length > 0) return assertUsableApiKey(hit, 'llm-pi-ai', ref)
|
||||
throw new LlmError(
|
||||
`llm-pi-ai: no credential for provider route "${provider}"; its profile resolves ${ref}, which is not`
|
||||
+ ` set — store ${ref} through the credentials service (the web Models page writes it) or export it,`
|
||||
@@ -118,13 +171,46 @@ export function apply(ctx: Context, config: Config): void {
|
||||
})
|
||||
// The full installed catalog is configurable from the moment the plugin
|
||||
// mounts — dormant or not — so configuration surfaces can offer every
|
||||
// pi-ai provider before any route exists.
|
||||
ctx.llm.registerConfigurableProviders(getBuiltinProviders().map(provider => ({
|
||||
provider,
|
||||
displayName: provider,
|
||||
settingsNs: NS,
|
||||
settingsPath: ['providers', provider],
|
||||
})))
|
||||
// pi-ai provider before any route exists. Hand-declared routes join it as
|
||||
// profiles appear, and leave with them.
|
||||
let directory: DirectoryRegistrationHandle | undefined
|
||||
let directoryFacts: unknown
|
||||
const ensureDirectory = (): void => {
|
||||
const entries = directoryEntries(profiles())
|
||||
if (deepEqualJson(entries, directoryFacts)) return
|
||||
// Atomic replace, never dispose-then-register: a route another adapter
|
||||
// family already declares (a profile keyed `deepseek-official`) would
|
||||
// otherwise leave this plugin's whole directory withdrawn and the Models
|
||||
// page empty. The candidate set is validated first, so a collision keeps
|
||||
// the previous entries serving and only costs a diagnostic.
|
||||
if (directory === undefined) {
|
||||
directory = ctx.llm.registerConfigurableProviders(entries)
|
||||
} else {
|
||||
directory.replace(entries)
|
||||
}
|
||||
directoryFacts = entries
|
||||
}
|
||||
ensureDirectory()
|
||||
/**
|
||||
* The credential a named route already resolves, for an interrogation whose
|
||||
* draft carries none. A route being declared for the first time names no
|
||||
* profile yet, and a profile that names no credential defers to pi-ai's own
|
||||
* discovery, so both answer `undefined` and the endpoint is asked
|
||||
* unauthenticated — the same posture a request to that route would take.
|
||||
*/
|
||||
const storedApiKey = async (provider: string | undefined): Promise<string | undefined> => {
|
||||
if (provider === undefined) return undefined
|
||||
const profile = profiles().get(provider)
|
||||
if (profile === undefined) return undefined
|
||||
return resolveApiKey(provider, profile)
|
||||
}
|
||||
// Interrogating an endpoint is a configuration-time action over a draft, so
|
||||
// it is offered for the whole namespace rather than per route: the provider
|
||||
// a surface is adding does not exist yet. The draft is the whole request
|
||||
// except the credential: a configuration surface edits a redacted descriptor
|
||||
// and never holds a stored secret, so an already-configured route supplies
|
||||
// its own here rather than being interrogated unauthenticated.
|
||||
ctx.llm.registerModelDiscovery(NS, request => discoverModels(request, () => storedApiKey(request.provider)))
|
||||
// Route effects bind to this apply fiber via the stable `ctx` reference,
|
||||
// even when a swap runs inside the scoped settings callback below. A bare
|
||||
// mount (zero routes) is the dormant posture: nothing registers until a
|
||||
@@ -157,9 +243,37 @@ export function apply(ctx: Context, config: Config): void {
|
||||
ensureRegistrationFacts()
|
||||
|
||||
installSettingsSection(ctx, NS, Config, config, {
|
||||
// Refuse an unserviceable section where it is written: without this a
|
||||
// schema-valid profile the adapter cannot serve would be stored and then
|
||||
// silently disable every route in this namespace.
|
||||
validate: assertServiceable,
|
||||
setSource: (source) => {
|
||||
current = source
|
||||
},
|
||||
onChange: ensureRegistrationFacts,
|
||||
onChange: () => {
|
||||
// Named here rather than left to the settings watcher: `assertServiceable`
|
||||
// cannot see the llm registry, so a profile claiming a route another
|
||||
// adapter family owns is stored successfully and only fails at this swap.
|
||||
// Without its own diagnostic that refusal reaches the operator as a
|
||||
// generic "settings: watcher failed", naming neither the route nor why it
|
||||
// is not serving. The previous routes keep serving either way.
|
||||
try {
|
||||
ensureRegistrationFacts()
|
||||
} catch (error) {
|
||||
ctx.logger.error('llm-pi-ai: keeping the previously registered routes after a refused update')
|
||||
ctx.logger.error(error)
|
||||
}
|
||||
// The directory follows the profiles the registry accepted, so a route
|
||||
// that failed to register is not advertised as configurable. A refused
|
||||
// directory swap is contained here for the same reason the registry's
|
||||
// is: the previous entries keep serving, and `directoryFacts` stays put
|
||||
// so returning to a working configuration re-applies.
|
||||
try {
|
||||
ensureDirectory()
|
||||
} catch (error) {
|
||||
ctx.logger.error('llm-pi-ai: keeping the previous configurable-provider directory after a refused update')
|
||||
ctx.logger.error(error)
|
||||
}
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
191
packages/llm/llm-pi-ai/src/provider.ts
Normal file
191
packages/llm/llm-pi-ai/src/provider.ts
Normal file
@@ -0,0 +1,191 @@
|
||||
/**
|
||||
* Construction of the pi-ai `Provider` that one configured route registers into
|
||||
* the adapter's `Models` collection.
|
||||
*
|
||||
* Two constructions, one decision: a route the installed catalog ships, whose
|
||||
* profile does not override the wire protocol, **reuses that catalog provider**
|
||||
* with its models replaced — the catalog provider owns API implementations this
|
||||
* package cannot reconstruct (Bedrock loads its Smithy module through a
|
||||
* separate entry point), so rebuilding it from parts would silently narrow
|
||||
* which providers work. Every other route — one pi-ai has never heard of, or a
|
||||
* catalog route pointed at a different protocol — is built by `createProvider`
|
||||
* over the protocol table below.
|
||||
*
|
||||
* Credentials never reach this module's storage: the harness resolves a route's
|
||||
* key through `ctx.credentials` before the request enters pi-ai and hands it
|
||||
* over as a stream option, which `Models` presents to `resolve()` as the
|
||||
* credential key.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/provider
|
||||
*/
|
||||
|
||||
import { createProvider } from '@earendil-works/pi-ai'
|
||||
import type { Api, ApiKeyAuth, Model, Provider, ProviderStreams } from '@earendil-works/pi-ai'
|
||||
import { anthropicMessagesApi } from '@earendil-works/pi-ai/api/anthropic-messages.lazy'
|
||||
import { openAICompletionsApi } from '@earendil-works/pi-ai/api/openai-completions.lazy'
|
||||
import { openAIResponsesApi } from '@earendil-works/pi-ai/api/openai-responses.lazy'
|
||||
import { catalogProvider } from './catalog.ts'
|
||||
|
||||
/**
|
||||
* Wire protocols a configured route may name, mapped to pi-ai's lazily loaded
|
||||
* implementations. Each entry is the factory that pi-ai's matching provider
|
||||
* factory uses, so a hand-declared route reaches exactly the implementation a
|
||||
* catalog route would.
|
||||
*
|
||||
* The table is deliberately narrow: the protocols a hand-declared route
|
||||
* actually reaches for today, each completely describable with a key, an
|
||||
* endpoint, and headers. Bedrock signs with SigV4 over AWS credentials and a
|
||||
* region, Vertex needs a project, a location, and application-default
|
||||
* credentials, Azure needs provider environment plus an api-version, and Codex
|
||||
* authenticates through OAuth — none of which this configuration shape can
|
||||
* express, so offering them would hand back a provider that cannot
|
||||
* authenticate. The remainder are absent for want of a consumer rather than a
|
||||
* blocker: each is one line here once a deployment needs it. Catalog routes
|
||||
* still reach every protocol through their own provider; only an explicit
|
||||
* override is refused.
|
||||
*/
|
||||
const PROTOCOLS: Readonly<Record<string, () => ProviderStreams>> = {
|
||||
'openai-completions': openAICompletionsApi,
|
||||
'openai-responses': openAIResponsesApi,
|
||||
'anthropic-messages': anthropicMessagesApi,
|
||||
}
|
||||
|
||||
/**
|
||||
* Every wire protocol a configured route may name, most-reached first. The
|
||||
* order is the table's and therefore stable; a configuration surface offering
|
||||
* a choice presents the first as its default, which is why the protocol a
|
||||
* hand-declared gateway most often speaks — and the one endpoint interrogation
|
||||
* can read — leads.
|
||||
* @returns the supported protocol identifiers.
|
||||
*/
|
||||
export function supportedProtocols(): readonly string[] {
|
||||
return Object.keys(PROTOCOLS)
|
||||
}
|
||||
|
||||
/**
|
||||
* Api-key auth for a route the harness authenticates itself. `Models` calls
|
||||
* this after the adapter has already resolved the route's credential, so a
|
||||
* missing key here is not this layer's failure: a named-but-unresolvable
|
||||
* reference has already failed the request with `MISSING_CREDENTIAL`, and a
|
||||
* route naming no credential at all is deliberately unauthenticated. Reporting
|
||||
* it as configured hands the decision to the protocol, which is where the
|
||||
* requirement actually lives — pi-ai's OpenAI-compatible implementation, for
|
||||
* one, still insists on a key or an `Authorization` header of its own.
|
||||
* @param name - display name used as the resolution's status label.
|
||||
* @returns the api-key auth for a harness-authenticated route.
|
||||
*/
|
||||
function harnessApiKeyAuth(name: string): ApiKeyAuth {
|
||||
return {
|
||||
name,
|
||||
resolve: ({ credential }) => Promise.resolve({
|
||||
auth: credential?.key === undefined ? {} : { apiKey: credential.key },
|
||||
source: name,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
/** The resolved route facts provider construction reads. */
|
||||
export interface ProviderSpec {
|
||||
/** Provider route key; also the `Models` collection key and each model's `provider`. */
|
||||
provider: string
|
||||
/** Display name for selectors and status labels. */
|
||||
displayName: string
|
||||
/** Wire protocol override; absent means each model keeps its catalog protocol. */
|
||||
api?: string
|
||||
/** Endpoint override already applied to {@link models}; kept for provider-level display. */
|
||||
baseURL?: string
|
||||
/** The route's materialized models, in configuration order. */
|
||||
models: readonly Model<Api>[]
|
||||
/**
|
||||
* Whether the profile names a credential — a literal key or a reference.
|
||||
* Only that decides whether {@link routeAuth} adds the harness's own api-key
|
||||
* method to a catalog provider that offers none; the key itself still arrives
|
||||
* per request, never at construction.
|
||||
*/
|
||||
namesCredential: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* The auth one route resolves its credential through.
|
||||
*
|
||||
* A catalog route keeps the installed provider's own auth, which is what
|
||||
* preserves provider-native ambient discovery for a profile naming no
|
||||
* credential. That holds even when the profile repoints the protocol: which
|
||||
* environment a provider reads is a property of the provider, not of the wire
|
||||
* format its models speak.
|
||||
*
|
||||
* The single addition covers a catalog provider that offers no api-key method
|
||||
* at all. pi-ai resolves a request's `apiKey` override only when the provider
|
||||
* declares one (`resolveProviderAuth` checks `provider.auth.apiKey` before
|
||||
* honouring the override), so an OAuth-only provider — `openai-codex` is the
|
||||
* one the installed catalog ships — would refuse a profile's explicit key with
|
||||
* `Provider is not configured` before any request went out. Adding the harness
|
||||
* method beside the provider's own restores that route. A keyless profile adds
|
||||
* nothing and still reports the honest refusal, because this adapter resolves
|
||||
* credentials through its own seam and holds no OAuth store to fall back on.
|
||||
* @param spec - the resolved route facts.
|
||||
* @param catalog - the installed catalog provider, when pi-ai ships one.
|
||||
* @returns the auth to construct this route's provider with.
|
||||
*/
|
||||
function routeAuth(spec: ProviderSpec, catalog: Provider | undefined): Provider['auth'] {
|
||||
if (catalog === undefined) return { apiKey: harnessApiKeyAuth(spec.displayName) }
|
||||
if (catalog.auth.apiKey !== undefined || !spec.namesCredential) return catalog.auth
|
||||
return { ...catalog.auth, apiKey: harnessApiKeyAuth(spec.displayName) }
|
||||
}
|
||||
|
||||
/**
|
||||
* Reuse an installed catalog provider with this route's models and identity.
|
||||
* Model dispatch stays with the catalog provider, so its API implementations,
|
||||
* compatibility quirks, and ambient credential discovery are preserved exactly.
|
||||
* Catalog-owned dynamic refresh is dropped: this route's catalog is the
|
||||
* settings document, and a background refresh would contradict it.
|
||||
*/
|
||||
function reuseCatalogProvider(base: Provider, spec: ProviderSpec): Provider {
|
||||
// Provider-level `baseUrl` is display metadata: pi-ai routes every request
|
||||
// through `Model.baseUrl`, which model resolution has already overridden.
|
||||
const baseUrl = spec.baseURL ?? base.baseUrl
|
||||
return {
|
||||
id: spec.provider,
|
||||
name: spec.displayName,
|
||||
...baseUrl === undefined ? {} : { baseUrl },
|
||||
auth: routeAuth(spec, base),
|
||||
getModels: () => spec.models,
|
||||
// Delegated rather than copied: the catalog provider stays the receiver, so
|
||||
// an implementation holding state on itself keeps working.
|
||||
stream: (model, context, options) => base.stream(model, context, options),
|
||||
streamSimple: (model, context, options) => base.streamSimple(model, context, options),
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the pi-ai provider for one resolved route.
|
||||
* @param spec - the resolved route facts.
|
||||
* @returns the provider to register in the adapter's `Models` collection.
|
||||
* @throws Error when the route names a wire protocol this build cannot serve.
|
||||
*/
|
||||
export function buildProvider(spec: ProviderSpec): Provider {
|
||||
const catalog = catalogProvider(spec.provider)
|
||||
// A catalog route keeping its catalog protocol reuses the catalog provider;
|
||||
// an explicit protocol means the deployment is repointing the route at a
|
||||
// different wire format, which only the protocol table can serve.
|
||||
if (catalog !== undefined && spec.api === undefined) return reuseCatalogProvider(catalog, spec)
|
||||
|
||||
// Every model on this path carries the route's protocol: model resolution
|
||||
// requires one for a route the catalog cannot default, and an explicit one
|
||||
// replaces each catalog model's own. So the route has a single API.
|
||||
const factory = spec.api === undefined ? undefined : PROTOCOLS[spec.api]
|
||||
if (factory === undefined) {
|
||||
throw new Error(
|
||||
`llm-pi-ai: provider "${spec.provider}" names api "${spec.api}", which this build cannot serve;`
|
||||
+ ` supported protocols are ${supportedProtocols().join(', ')}`,
|
||||
)
|
||||
}
|
||||
return createProvider({
|
||||
id: spec.provider,
|
||||
name: spec.displayName,
|
||||
...spec.baseURL === undefined ? {} : { baseUrl: spec.baseURL },
|
||||
auth: routeAuth(spec, catalog),
|
||||
models: spec.models,
|
||||
api: factory(),
|
||||
})
|
||||
}
|
||||
@@ -100,7 +100,7 @@ describe('PiAiAdapter provider routing', () => {
|
||||
})
|
||||
})
|
||||
|
||||
it('uses a dynamic request effort and rejects unsupported efforts before network I/O', async () => {
|
||||
it('uses a dynamic request effort and reports unsupported efforts before network I/O', async () => {
|
||||
const server = await mockServer([{ events: textEvents }, { events: textEvents }])
|
||||
const ctx = await harness(server.url, { reasoning: 'max' })
|
||||
|
||||
@@ -119,11 +119,15 @@ describe('PiAiAdapter provider routing', () => {
|
||||
expect(server.requests[1]).toMatchObject({ thinking: { type: 'disabled' } })
|
||||
expect(server.requests[1]).not.toHaveProperty('reasoning_effort')
|
||||
|
||||
await expect(assemble(ctx, {
|
||||
const unsupported = await assemble(ctx, {
|
||||
model: 'deepseek-v4-flash',
|
||||
reasoningEffort: ReasoningEffortId('xhigh'),
|
||||
messages: [],
|
||||
})).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' })
|
||||
})
|
||||
expect(unsupported.finish).toMatchObject({
|
||||
kind: 'error',
|
||||
failure: { code: 'UNSUPPORTED_REASONING_EFFORT' },
|
||||
})
|
||||
expect(server.requests).toHaveLength(2)
|
||||
})
|
||||
|
||||
@@ -140,19 +144,36 @@ describe('PiAiAdapter provider routing', () => {
|
||||
expect(result.message.content).toEqual([{ type: 'text', text: 'hello' }])
|
||||
})
|
||||
|
||||
it('rejects stop sequences rather than silently ignoring them', async () => {
|
||||
it('names a route by its displayName, and by its own key once the profiles drop it', () => {
|
||||
const adapter = adapterOf({ 'acme-gateway': {
|
||||
apiKey: 'k',
|
||||
displayName: 'Acme Gateway',
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://acme.test/v1',
|
||||
models: [{ id: 'acme-large' }],
|
||||
} })
|
||||
expect(adapter.providerInfo('acme-gateway')).toEqual({ id: 'acme-gateway', name: 'Acme Gateway' })
|
||||
|
||||
// The registry and the profiles can disagree for a moment: a refused
|
||||
// registration swap leaves the previous routes serving while resolution
|
||||
// has already moved on, so a selector may ask about a route the current
|
||||
// profiles no longer describe. It gets the key rather than nothing.
|
||||
expect(adapter.providerInfo('departed')).toEqual({ id: 'departed', name: 'departed' })
|
||||
})
|
||||
|
||||
it('reports unsupported stop sequences rather than silently ignoring them', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness(server.url)
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [], stop: ['END'] }))
|
||||
.rejects.toMatchObject({ code: 'UNSUPPORTED_OPTION' })
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [], stop: ['END'] })
|
||||
expect(result.finish).toMatchObject({ kind: 'error', failure: { code: 'UNSUPPORTED_OPTION' } })
|
||||
expect(server.requests).toEqual([])
|
||||
})
|
||||
|
||||
it('rejects unknown catalog models before network I/O', async () => {
|
||||
it('reports unknown catalog models before network I/O', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness(server.url)
|
||||
await expect(assemble(ctx, { model: 'not-in-the-catalog', messages: [] }))
|
||||
.rejects.toMatchObject({ code: 'UNKNOWN_MODEL' })
|
||||
const result = await assemble(ctx, { model: 'not-in-the-catalog', messages: [] })
|
||||
expect(result.finish).toMatchObject({ kind: 'error', failure: { code: 'UNKNOWN_MODEL' } })
|
||||
expect(server.requests).toEqual([])
|
||||
})
|
||||
|
||||
@@ -308,8 +329,8 @@ describe('PiAiAdapter provider routing', () => {
|
||||
const server = await mockServer([{ events: textEvents, delayMs: 200 }])
|
||||
const ctx = await harness(server.url, { streamIdleTimeoutMs: 20 })
|
||||
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toMatchObject({ code: 'TIMEOUT' })
|
||||
const result = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(result.finish).toMatchObject({ kind: 'error', failure: { code: 'TIMEOUT' } })
|
||||
await Promise.race([
|
||||
server.responseClosed,
|
||||
new Promise<never>((_resolve, reject) => {
|
||||
@@ -407,12 +428,11 @@ describe('provider profile lifecycle', () => {
|
||||
ReasoningEffortId('xhigh'),
|
||||
ReasoningEffortId('max'),
|
||||
])
|
||||
await expect(ctx.llm.resolveModelInfo('openai', 'gpt-4.1'))
|
||||
.resolves.toMatchObject({
|
||||
reasoning: {
|
||||
efforts: [{ id: ReasoningEffortId('off'), name: 'Off' }],
|
||||
},
|
||||
})
|
||||
// A catalog model without reasoning is the same case as a hand-declared
|
||||
// one: pi-ai reports the single level `off`, which translates to omitting
|
||||
// the reasoning option — exactly what naming no effort already does. The
|
||||
// capability is reported unavailable rather than offering that control.
|
||||
expect((await ctx.llm.resolveModelInfo('openai', 'gpt-4.1')).reasoning).toBeUndefined()
|
||||
})
|
||||
|
||||
it('uses a supported profile reasoning value as the model default and rejects an unsupported one', async () => {
|
||||
@@ -424,13 +444,24 @@ describe('provider profile lifecycle', () => {
|
||||
await expect(supported.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
|
||||
.resolves.toMatchObject({ reasoning: { defaultEffort: ReasoningEffortId('max') } })
|
||||
|
||||
// A profile level this model cannot take DESCRIBES as no default rather
|
||||
// than failing: resolveModelInfo builds the model catalog, and a catalog
|
||||
// that throws takes its whole provider out of every picker — one mis-set
|
||||
// field would hide every model on the route, including the ones that do
|
||||
// support the level. The request path below is where it is refused.
|
||||
const unsupported = new Context()
|
||||
await unsupported.plugin(LlmService)
|
||||
await unsupported.plugin(LlmPiAi, {
|
||||
providers: { deepseek: { reasoning: 'medium' } },
|
||||
})
|
||||
await expect(unsupported.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
|
||||
.rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' })
|
||||
const described = await unsupported.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash')
|
||||
expect(described.reasoning?.defaultEffort).toBeUndefined()
|
||||
expect(described.reasoning?.efforts.length).toBeGreaterThan(0)
|
||||
await expect(assemble(unsupported, {
|
||||
provider: 'deepseek', model: 'deepseek-v4-flash', messages: [],
|
||||
})).resolves.toMatchObject({
|
||||
finish: { kind: 'error', failure: { code: 'UNSUPPORTED_REASONING_EFFORT' } },
|
||||
})
|
||||
|
||||
const disabled = new Context()
|
||||
await disabled.plugin(LlmService)
|
||||
@@ -465,19 +496,23 @@ describe('provider profile lifecycle', () => {
|
||||
vi.stubEnv('DEEPSEEK_API_KEY', 'ambient-key')
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
const ctx = await harness(server.url, { apiKey: undefined, apiKeyEnv: 'PI_CUSTOM_REF_KEY' })
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toMatchObject({ code: 'MISSING_CREDENTIAL' })
|
||||
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toThrow(/provider route "deepseek".*PI_CUSTOM_REF_KEY/s)
|
||||
const first = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(first.finish).toMatchObject({ kind: 'error', failure: { code: 'MISSING_CREDENTIAL' } })
|
||||
const second = await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(second.finish.kind).toBe('error')
|
||||
if (second.finish.kind !== 'error') throw new Error('expected an error finish')
|
||||
expect(second.finish.failure.message).toMatch(/provider route "deepseek".*PI_CUSTOM_REF_KEY/s)
|
||||
expect(server.requests).toHaveLength(0)
|
||||
})
|
||||
|
||||
it('validates empty, unknown, legacy-shaped, and explicitly blank profiles', () => {
|
||||
it('validates empty, underspecified, legacy-shaped, and explicitly blank profiles', () => {
|
||||
// Empty and omitted dicts are the dormant zero-route posture, not errors.
|
||||
expect(resolveProfiles({}).size).toBe(0)
|
||||
expect(resolveProfiles(undefined).size).toBe(0)
|
||||
expect(() => resolveProfiles({ '': {} })).toThrow(/non-empty/)
|
||||
expect(() => resolveProfiles({ 'not-real': {} })).toThrow(/unknown/)
|
||||
// A route the installed catalog does not ship is allowed, but it has no
|
||||
// defaults to fall back on: it must describe its own models.
|
||||
expect(() => resolveProfiles({ 'not-real': {} })).toThrow(/resolves no models/)
|
||||
// The pre-release array shape and its per-profile provider field fail
|
||||
// loud with migration directions instead of half-working.
|
||||
expect(() => resolveProfiles([{ provider: 'openai' }] as never)).toThrow(/dict keyed by provider/)
|
||||
|
||||
585
packages/llm/llm-pi-ai/tests/catalog.spec.ts
Normal file
585
packages/llm/llm-pi-ai/tests/catalog.spec.ts
Normal file
@@ -0,0 +1,585 @@
|
||||
import { mkdtemp, rm, writeFile } from 'node:fs/promises'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService, { createUserMessage } from '@deepseek-ai/dsh-llm'
|
||||
import type { StreamChunk } from '@deepseek-ai/dsh-llm'
|
||||
import SettingsLocal from '@deepseek-ai/dsh-settings-local'
|
||||
import { settingsNamespace } from '@deepseek-ai/dsh-settings'
|
||||
import * as LlmPiAi from '@deepseek-ai/dsh-llm-pi-ai'
|
||||
import { PiAiAdapter } from '@deepseek-ai/dsh-llm-pi-ai'
|
||||
import { getBuiltinModels } from '@earendil-works/pi-ai/providers/all'
|
||||
import { createModels } from '@earendil-works/pi-ai'
|
||||
import type { Api, Model, Provider } from '@earendil-works/pi-ai'
|
||||
import { resolveProfiles } from '../src/config.ts'
|
||||
import { buildProvider, supportedProtocols } from '../src/provider.ts'
|
||||
import { assemble } from './assemble.ts'
|
||||
import { closeMockServers, mockServer, textEvents } from './mock-server.ts'
|
||||
|
||||
const homes: string[] = []
|
||||
|
||||
afterEach(async () => {
|
||||
await closeMockServers()
|
||||
await Promise.all(homes.splice(0).map(dir => rm(dir, { recursive: true, force: true })))
|
||||
})
|
||||
|
||||
/** A throwaway $DSH_HOME with an empty settings document. */
|
||||
async function home(): Promise<string> {
|
||||
const dir = await mkdtemp(join(tmpdir(), 'dsh-pi-catalog-'))
|
||||
homes.push(dir)
|
||||
await writeFile(join(dir, 'settings.yaml'), '')
|
||||
return dir
|
||||
}
|
||||
|
||||
/** The dormant composition plus a real settings service, as the product mounts it. */
|
||||
async function bootWithSettings(dir: string, config: LlmPiAi.Config): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(SettingsLocal, { path: join(dir, 'settings.yaml'), watch: false })
|
||||
await ctx.plugin(LlmPiAi, config)
|
||||
return ctx
|
||||
}
|
||||
|
||||
/** A complete hand-declared route: nothing about it exists in pi-ai's catalog. */
|
||||
function gateway(baseURL: string, overrides: Record<string, unknown> = {}): LlmPiAi.Config {
|
||||
return {
|
||||
providers: {
|
||||
'acme-gateway': {
|
||||
apiKey: 'gw-key',
|
||||
displayName: 'Acme Gateway',
|
||||
api: 'openai-completions',
|
||||
baseURL,
|
||||
models: [{ id: 'acme-large', name: 'Acme Large', contextWindow: 65_536, maxTokens: 4096 }],
|
||||
...overrides,
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
async function harness(config: LlmPiAi.Config): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(LlmPiAi, config)
|
||||
return ctx
|
||||
}
|
||||
|
||||
describe('hand-declared providers', () => {
|
||||
it('serves a route pi-ai has never heard of from its own declaration', async () => {
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
const ctx = await harness(gateway(`${server.url}/v1`))
|
||||
|
||||
const result = await assemble(ctx, {
|
||||
provider: 'acme-gateway',
|
||||
model: 'acme-large',
|
||||
messages: [createUserMessage({
|
||||
content: [{ type: 'text', text: 'hi' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
})],
|
||||
})
|
||||
|
||||
expect(result.message.content).toEqual([{ type: 'text', text: 'hello' }])
|
||||
expect(result.finish).toEqual({ kind: 'stop' })
|
||||
expect(server.paths).toEqual(['/v1/chat/completions'])
|
||||
expect(server.headers[0]?.authorization).toBe('Bearer gw-key')
|
||||
})
|
||||
|
||||
it('lists and resolves the declared models rather than a catalog', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness(gateway(`${server.url}/v1`))
|
||||
|
||||
expect(await ctx.llm.listModels('acme-gateway')).toEqual([
|
||||
{ provider: 'acme-gateway', id: 'acme-large', name: 'Acme Large', inputModalities: ['text'] },
|
||||
])
|
||||
const info = await ctx.llm.resolveModelInfo('acme-gateway', 'acme-large')
|
||||
expect(info).toMatchObject({
|
||||
provider: 'acme-gateway',
|
||||
id: 'acme-large',
|
||||
name: 'Acme Large',
|
||||
context: { contextWindow: 65_536 },
|
||||
defaultMaxTokens: 4096,
|
||||
})
|
||||
})
|
||||
|
||||
it('offers no reasoning control it could not honour', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness(gateway(`${server.url}/v1`))
|
||||
|
||||
// pi-ai reports a model with no reasoning metadata as supporting the single
|
||||
// level `off`, but `off` is translated to *omitting* the reasoning option —
|
||||
// byte-for-byte the same request as naming no effort — so a provider whose
|
||||
// own default is to think would keep thinking with `off` selected. The
|
||||
// capability is reported unavailable instead of offering that control.
|
||||
expect((await ctx.llm.resolveModelInfo('acme-gateway', 'acme-large')).reasoning).toBeUndefined()
|
||||
|
||||
// A catalog route is unaffected: its models carry the metadata that makes
|
||||
// `off` actually disable thinking.
|
||||
const withCatalog = await harness({ providers: { deepseek: { apiKey: 'k', baseURL: server.url } } })
|
||||
const [catalogModel] = getBuiltinModels('deepseek')
|
||||
if (catalogModel === undefined) throw new Error('the installed catalog ships no deepseek model')
|
||||
expect((await withCatalog.llm.resolveModelInfo('deepseek', catalogModel.id)).reasoning?.efforts.map(e => e.id))
|
||||
.toContain('off')
|
||||
})
|
||||
|
||||
it('joins the configurable-provider directory so a settings surface can reach it', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness(gateway(`${server.url}/v1`))
|
||||
const directory = ctx.llm.listConfigurableProviders()
|
||||
|
||||
expect(directory).toContainEqual({
|
||||
provider: 'acme-gateway',
|
||||
displayName: 'Acme Gateway',
|
||||
settingsNs: 'llm-pi-ai',
|
||||
settingsPath: ['providers', 'acme-gateway'],
|
||||
// Nothing in the installed catalog answers for this route, which is what
|
||||
// configuration surfaces mark as a route this deployment declared.
|
||||
declared: true,
|
||||
})
|
||||
// Membership of the catalog, not of the settings document: a shipped
|
||||
// provider carries a stored profile the moment anyone corrects it.
|
||||
expect(directory.filter(entry => entry.declared).map(entry => entry.provider))
|
||||
.toEqual(['acme-gateway'])
|
||||
expect(directory.find(entry => entry.provider === 'deepseek')?.declared).toBe(false)
|
||||
})
|
||||
|
||||
it('sizes a model the catalog cannot describe from the route\u2019s own fallbacks', () => {
|
||||
const resolved = resolveProfiles({
|
||||
'acme-gateway': {
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://acme.test',
|
||||
// A listing endpoint that discloses nothing but ids still yields a
|
||||
// serviceable route.
|
||||
models: [{ id: 'bare' }, { id: 'sized', contextWindow: 8192, maxTokens: 512 }],
|
||||
},
|
||||
'tuned-gateway': {
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://tuned.test',
|
||||
defaultContextWindow: 4096,
|
||||
defaultMaxTokens: 256,
|
||||
models: [{ id: 'bare' }],
|
||||
},
|
||||
})
|
||||
const modelsOf = (route: string): readonly { id: string; contextWindow: number; maxTokens: number }[] =>
|
||||
resolved.get(route)?.piProvider.getModels() ?? []
|
||||
|
||||
expect(modelsOf('acme-gateway')).toMatchObject([
|
||||
{ id: 'bare', contextWindow: 262_144, maxTokens: 32_768 },
|
||||
{ id: 'sized', contextWindow: 8192, maxTokens: 512 },
|
||||
])
|
||||
// The fallback is a guess, so a deployment whose gateway serves smaller
|
||||
// models corrects it once for the whole route.
|
||||
expect(modelsOf('tuned-gateway')).toMatchObject([{ id: 'bare', contextWindow: 4096, maxTokens: 256 }])
|
||||
// Only an explicitly configured cap is a request default; a fallback is
|
||||
// the model's capability and stops there.
|
||||
expect(resolved.get('acme-gateway')?.configuredMaxTokens.get('bare')).toBeUndefined()
|
||||
expect(resolved.get('acme-gateway')?.configuredMaxTokens.get('sized')).toBe(512)
|
||||
})
|
||||
|
||||
it('rejects a model the route cannot identify', () => {
|
||||
const declare = (model: LlmPiAi.PiAiModelProfile): (() => unknown) =>
|
||||
() => resolveProfiles({ 'acme-gateway': { api: 'openai-completions', baseURL: 'https://acme.test', models: [model] } })
|
||||
|
||||
expect(declare({ id: '' })).toThrow(/empty id/)
|
||||
expect(() => resolveProfiles({
|
||||
'acme-gateway': {
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://acme.test',
|
||||
models: [{ id: 'dup', contextWindow: 1, maxTokens: 1 }, { id: 'dup', contextWindow: 2, maxTokens: 2 }],
|
||||
},
|
||||
})).toThrow(/more than once/)
|
||||
})
|
||||
|
||||
it('rejects a declaration that names no wire protocol or endpoint', () => {
|
||||
expect(() => resolveProfiles({
|
||||
'acme-gateway': { baseURL: 'https://acme.test', models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }] },
|
||||
})).toThrow(/needs an api/)
|
||||
expect(() => resolveProfiles({
|
||||
'acme-gateway': { api: 'openai-completions', models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }] },
|
||||
})).toThrow(/needs a baseURL/)
|
||||
})
|
||||
|
||||
it.each(['bedrock-converse-stream', 'google-vertex', 'azure-openai-responses', 'openai-codex-responses'])(
|
||||
'refuses %s, whose authentication a profile cannot express',
|
||||
(api) => {
|
||||
// These need SigV4 credentials and a region, a project plus ADC, provider
|
||||
// environment and an api-version, or OAuth — none of which a key, an
|
||||
// endpoint, and headers can carry, so a route naming one would be built
|
||||
// unable to authenticate.
|
||||
expect(supportedProtocols()).not.toContain(api)
|
||||
expect(() => buildProvider({ provider: 'acme-gateway', displayName: 'Acme', api, models: [], namesCredential: true }))
|
||||
.toThrow(/cannot serve; supported protocols are/)
|
||||
},
|
||||
)
|
||||
|
||||
it('rejects a protocol this build cannot serve, and a route that names none', () => {
|
||||
const spec = { provider: 'acme-gateway', displayName: 'Acme Gateway', models: [], namesCredential: true }
|
||||
expect(() => buildProvider({ ...spec, api: 'quantum-telepathy' }))
|
||||
.toThrow(/cannot serve; supported protocols are/)
|
||||
expect(() => buildProvider(spec)).toThrow(/cannot serve; supported protocols are/)
|
||||
})
|
||||
|
||||
it('leaves an unauthenticated route to its protocol rather than inventing a credential', async () => {
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
// Naming no credential is the deliberately unauthenticated posture — a
|
||||
// named reference that resolved to nothing would have failed with
|
||||
// MISSING_CREDENTIAL long before this point. The route resolves as
|
||||
// configured and the protocol decides: pi-ai's OpenAI-compatible
|
||||
// implementation wants a key or an Authorization header of its own, and
|
||||
// says so instead of the harness guessing a placeholder.
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
'local-llm': {
|
||||
api: 'openai-completions',
|
||||
baseURL: `${server.url}/v1`,
|
||||
models: [{ id: 'qwen3', contextWindow: 32_768, maxTokens: 2048 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
const result = await assemble(ctx, { provider: 'local-llm', model: 'qwen3', messages: [] })
|
||||
expect(result.finish).toMatchObject({
|
||||
kind: 'error',
|
||||
failure: { message: 'No API key for provider: local-llm' },
|
||||
})
|
||||
expect(server.requests).toHaveLength(0)
|
||||
})
|
||||
|
||||
it('authenticates an unauthenticated route through a configured header', async () => {
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
'local-llm': {
|
||||
api: 'openai-completions',
|
||||
baseURL: `${server.url}/v1`,
|
||||
headers: { Authorization: 'Bearer local' },
|
||||
models: [{ id: 'qwen3', contextWindow: 32_768, maxTokens: 2048 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
const result = await assemble(ctx, { provider: 'local-llm', model: 'qwen3', messages: [] })
|
||||
expect(result.finish).toEqual({ kind: 'stop' })
|
||||
expect(server.headers[0]?.authorization).toBe('Bearer local')
|
||||
})
|
||||
|
||||
it('rejects a capacity that is not a positive integer', () => {
|
||||
const declare = (model: LlmPiAi.PiAiModelProfile): (() => unknown) =>
|
||||
() => resolveProfiles({ 'acme-gateway': { api: 'openai-completions', baseURL: 'https://acme.test', models: [model] } })
|
||||
|
||||
expect(declare({ id: 'm', contextWindow: 0, maxTokens: 1 })).toThrow(/contextWindow must be a positive integer/)
|
||||
expect(declare({ id: 'm', contextWindow: 1.5, maxTokens: 1 })).toThrow(/contextWindow must be a positive integer/)
|
||||
expect(declare({ id: 'm', contextWindow: 1, maxTokens: 0 })).toThrow(/maxTokens must be a positive integer/)
|
||||
expect(declare({ id: 'm', contextWindow: 1, maxTokens: 1.5 })).toThrow(/maxTokens must be a positive integer/)
|
||||
})
|
||||
|
||||
it('names the route key when no displayName is configured', () => {
|
||||
const resolved = resolveProfiles({
|
||||
'acme-gateway': {
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://acme.test',
|
||||
models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }],
|
||||
},
|
||||
})
|
||||
expect(resolved.get('acme-gateway')?.displayName).toBe('acme-gateway')
|
||||
expect(() => resolveProfiles({ 'acme-gateway': { displayName: '' } })).toThrow(/empty displayName/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('catalog routes with per-model configuration', () => {
|
||||
it('serves the installed catalog untouched when the profile lists no models', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness({ providers: { deepseek: { apiKey: 'k', baseURL: server.url } } })
|
||||
|
||||
const listed = await ctx.llm.listModels('deepseek')
|
||||
expect(listed.map(model => model.id).sort())
|
||||
.toEqual(getBuiltinModels('deepseek').map(model => model.id).sort())
|
||||
})
|
||||
|
||||
it('overrides one catalog model field and defaults the rest from the catalog', async () => {
|
||||
const server = await mockServer([])
|
||||
const [catalogModel] = getBuiltinModels('deepseek')
|
||||
if (catalogModel === undefined) throw new Error('the installed catalog ships no deepseek model')
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
deepseek: {
|
||||
apiKey: 'k',
|
||||
baseURL: server.url,
|
||||
models: [{ id: catalogModel.id, contextWindow: 4096 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
const info = await ctx.llm.resolveModelInfo('deepseek', catalogModel.id)
|
||||
// The configured field wins and the name still comes from the catalog. The
|
||||
// catalog's own output cap is the model's capability, not a cap anyone
|
||||
// chose, so it must not arrive as the request default.
|
||||
expect(info.context).toEqual({ contextWindow: 4096 })
|
||||
expect(info.name).toBe(catalogModel.name)
|
||||
expect(info.defaultMaxTokens).toBeUndefined()
|
||||
// An explicit list replaces the catalog rather than adding to it.
|
||||
expect((await ctx.llm.listModels('deepseek')).map(model => model.id)).toEqual([catalogModel.id])
|
||||
})
|
||||
|
||||
it('materializes a request default only from a configured output cap', async () => {
|
||||
const server = await mockServer([])
|
||||
const [catalogModel] = getBuiltinModels('deepseek')
|
||||
if (catalogModel === undefined) throw new Error('the installed catalog ships no deepseek model')
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
deepseek: {
|
||||
apiKey: 'k',
|
||||
baseURL: server.url,
|
||||
models: [{ id: catalogModel.id, maxTokens: 4096 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
// Configuring the cap is the deployment choosing one, so it becomes the
|
||||
// default the seam materializes into requests that name none.
|
||||
expect((await ctx.llm.resolveModelInfo('deepseek', catalogModel.id)).defaultMaxTokens).toBe(4096)
|
||||
})
|
||||
|
||||
it('adds a model the installed catalog does not describe to a catalog route', async () => {
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
deepseek: {
|
||||
apiKey: 'k',
|
||||
baseURL: `${server.url}/v1`,
|
||||
models: [{ id: 'deepseek-preview', contextWindow: 200_000, maxTokens: 8192 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
const result = await assemble(ctx, { provider: 'deepseek', model: 'deepseek-preview', messages: [] })
|
||||
expect(result.finish).toEqual({ kind: 'stop' })
|
||||
// The catalog route keeps its catalog protocol, so the new model reaches
|
||||
// the same endpoint shape the shipped models use.
|
||||
expect(server.paths).toEqual(['/v1/chat/completions'])
|
||||
})
|
||||
|
||||
it('fails an unconfigured model id before any provider request', async () => {
|
||||
const server = await mockServer([])
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
deepseek: { apiKey: 'k', baseURL: server.url, models: [{ id: 'deepseek-preview', contextWindow: 1, maxTokens: 1 }] },
|
||||
},
|
||||
})
|
||||
|
||||
const result = await assemble(ctx, { provider: 'deepseek', model: 'not-configured', messages: [] })
|
||||
|
||||
expect(result.finish).toMatchObject({ kind: 'error', failure: { code: 'UNKNOWN_MODEL' } })
|
||||
expect(server.requests).toHaveLength(0)
|
||||
})
|
||||
|
||||
it('preserves catalog-only model metadata the profile cannot express', () => {
|
||||
// Some catalog models carry provider-required request headers; overriding a
|
||||
// capacity must not drop them, because configuration has no way to restate
|
||||
// them.
|
||||
const headered = (getBuiltinModels('nvidia') as { id: string; headers?: unknown }[])
|
||||
.find(model => model.headers !== undefined)
|
||||
if (headered === undefined) throw new Error('the installed catalog ships no nvidia model with headers')
|
||||
|
||||
const resolved = resolveProfiles({
|
||||
nvidia: { models: [{ id: headered.id, contextWindow: 4096 }] },
|
||||
})
|
||||
const [model] = resolved.get('nvidia')?.piProvider.getModels() ?? []
|
||||
expect(model?.headers).toEqual(headered.headers)
|
||||
expect(model?.contextWindow).toBe(4096)
|
||||
})
|
||||
|
||||
it('delegates both stream methods back to the reused catalog provider', async () => {
|
||||
const server = await mockServer([{ events: textEvents }, { events: textEvents }])
|
||||
const resolved = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${server.url}/v1` } })
|
||||
const built = resolved.get('deepseek')?.piProvider
|
||||
if (built === undefined) throw new Error('the deepseek route built no provider')
|
||||
const [model] = built.getModels()
|
||||
if (model === undefined) throw new Error('the deepseek route resolved no models')
|
||||
const context = { messages: [{ role: 'user' as const, content: 'hi', timestamp: 0 }] }
|
||||
|
||||
// `stream` is interface-required and unused by the harness adapter, which
|
||||
// only calls `streamSimple`; both must still reach the catalog provider.
|
||||
for await (const _event of built.stream(model, context, { apiKey: 'k' })) { /* drain */ }
|
||||
for await (const _event of built.streamSimple(model, context, { apiKey: 'k' })) { /* drain */ }
|
||||
|
||||
expect(server.paths).toEqual(['/v1/chat/completions', '/v1/chat/completions'])
|
||||
})
|
||||
|
||||
it('keeps each model its own endpoint when the catalog route declares none', () => {
|
||||
// `opencode` ships no provider-level endpoint: the address lives on every
|
||||
// catalog model, so the route resolves without any configured baseURL.
|
||||
const resolved = resolveProfiles({ opencode: {} })
|
||||
const models = resolved.get('opencode')?.piProvider.getModels() ?? []
|
||||
expect(models.length).toBeGreaterThan(0)
|
||||
expect(models.every(model => model.baseUrl.length > 0)).toBe(true)
|
||||
expect(resolved.get('opencode')?.piProvider.baseUrl).toBeUndefined()
|
||||
})
|
||||
|
||||
it('repoints a catalog route at another wire protocol without restating its endpoint', () => {
|
||||
const resolved = resolveProfiles({ openai: { api: 'openai-completions' } })
|
||||
const models = resolved.get('openai')?.piProvider.getModels() ?? []
|
||||
// The protocol changes for the whole route; each model keeps the catalog
|
||||
// endpoint it already had.
|
||||
expect(models.every(model => model.api === 'openai-completions')).toBe(true)
|
||||
expect(models.every(model => model.baseUrl === 'https://api.openai.com/v1')).toBe(true)
|
||||
})
|
||||
|
||||
it('repoints a catalog route at another wire protocol', async () => {
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
const ctx = await harness({
|
||||
providers: {
|
||||
// openai's catalog models speak the Responses API; naming the protocol
|
||||
// explicitly moves the whole route onto Chat Completions.
|
||||
openai: {
|
||||
apiKey: 'k',
|
||||
api: 'openai-completions',
|
||||
baseURL: `${server.url}/v1`,
|
||||
models: [{ id: 'gpt-4.1', contextWindow: 100_000, maxTokens: 4096 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
await assemble(ctx, { provider: 'openai', model: 'gpt-4.1', messages: [] })
|
||||
expect(server.paths).toEqual(['/v1/chat/completions'])
|
||||
})
|
||||
|
||||
it('keeps the catalog provider’s own auth when the route repoints its protocol', () => {
|
||||
// Which environment a provider reads is a property of the provider, not of
|
||||
// the wire format its models speak: naming an api must not cost a profile
|
||||
// its provider-native discovery.
|
||||
const resolved = resolveProfiles({ openai: { api: 'openai-completions' } })
|
||||
expect(resolved.get('openai')?.piProvider.auth.apiKey?.name).toBe('OpenAI API key')
|
||||
})
|
||||
|
||||
it('lets an OAuth-only catalog route authenticate with the key its profile names', async () => {
|
||||
// pi-ai honours a request's `apiKey` override only when the provider
|
||||
// declares an api-key method. `openai-codex` ships OAuth alone, so without
|
||||
// the harness method beside it the route refuses its own configured key as
|
||||
// `Provider is not configured` before any request goes out.
|
||||
const resolved = resolveProfiles({ 'openai-codex': { apiKey: 'codex-token' } })
|
||||
const provider = resolved.get('openai-codex')?.piProvider
|
||||
expect(provider?.auth.oauth).toBeDefined()
|
||||
const models = createModels()
|
||||
models.setProvider(provider as Provider)
|
||||
const model = provider?.getModels()[0] as Model<Api>
|
||||
const auth = await models.getAuth(model, { apiKey: 'codex-token' })
|
||||
expect(auth?.auth.apiKey).toBe('codex-token')
|
||||
})
|
||||
|
||||
it('leaves an OAuth-only catalog route unconfigured when its profile names no key', () => {
|
||||
// Nothing to add: this adapter resolves credentials through its own seam
|
||||
// and holds no OAuth store, so declaring the provider configured would
|
||||
// trade a truthful refusal for an endpoint's 401.
|
||||
const resolved = resolveProfiles({ 'openai-codex': {} })
|
||||
expect(resolved.get('openai-codex')?.piProvider.auth.apiKey).toBeUndefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('resolution snapshots', () => {
|
||||
it('finishes an in-flight request under the configuration it started with', async () => {
|
||||
const server = await mockServer([{ events: textEvents }])
|
||||
let current = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${server.url}/v1` } })
|
||||
let release: () => void = () => {}
|
||||
const held = new Promise<void>((resolve) => { release = resolve })
|
||||
const adapter = new PiAiAdapter({
|
||||
profiles: () => current,
|
||||
// Credential resolution is the real await inside a stream call, and the
|
||||
// window a configuration change has to land in.
|
||||
resolveApiKey: async () => { await held; return 'k' },
|
||||
})
|
||||
|
||||
const chunks: StreamChunk[] = []
|
||||
const inFlight = (async () => {
|
||||
for await (const chunk of adapter.stream({
|
||||
provider: 'deepseek',
|
||||
model: 'deepseek-v4-flash',
|
||||
messages: [],
|
||||
})) chunks.push(chunk)
|
||||
})()
|
||||
|
||||
// The route set changes while the request waits, and something else reads
|
||||
// the adapter meanwhile, which is what would rebuild a shared collection.
|
||||
current = resolveProfiles({ openai: { apiKey: 'k', baseURL: `${server.url}/v1` } })
|
||||
await expect(adapter.listModels('openai')).resolves.not.toHaveLength(0)
|
||||
release()
|
||||
await inFlight
|
||||
|
||||
// The in-flight request keeps its own snapshot: it reaches the endpoint it
|
||||
// resolved against instead of failing on a provider that no longer exists.
|
||||
expect(chunks.at(-1)).toMatchObject({ type: 'finish', reason: { kind: 'stop' } })
|
||||
expect(server.paths).toEqual(['/v1/chat/completions'])
|
||||
})
|
||||
|
||||
it('serves the next request from the new configuration', async () => {
|
||||
const first = await mockServer([{ events: textEvents }])
|
||||
const second = await mockServer([{ events: textEvents }])
|
||||
let current = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${first.url}/v1` } })
|
||||
const adapter = new PiAiAdapter({ profiles: () => current, resolveApiKey: () => Promise.resolve('k') })
|
||||
const drain = async (): Promise<void> => {
|
||||
for await (const _chunk of adapter.stream({
|
||||
provider: 'deepseek', model: 'deepseek-v4-flash', messages: [],
|
||||
})) { /* drain */ }
|
||||
}
|
||||
|
||||
await drain()
|
||||
current = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${second.url}/v1` } })
|
||||
await drain()
|
||||
|
||||
expect(first.paths).toHaveLength(1)
|
||||
expect(second.paths).toHaveLength(1)
|
||||
})
|
||||
})
|
||||
|
||||
describe('configurable-provider directory', () => {
|
||||
it('keeps the previous directory when a route collides with another adapter family', async () => {
|
||||
const dir = await home()
|
||||
const ctx = await bootWithSettings(dir, {})
|
||||
// Another adapter family owns this route id, exactly as llm-deepseek does.
|
||||
ctx.llm.registerConfigurableProviders([
|
||||
{ provider: 'deepseek-official', displayName: 'DeepSeek', settingsNs: 'llm-deepseek', settingsPath: [] },
|
||||
])
|
||||
const before = ctx.llm.listConfigurableProviders().length
|
||||
expect(before).toBeGreaterThan(30)
|
||||
|
||||
await ctx.settings.update(settingsNamespace('llm-pi-ai'), {
|
||||
providers: {
|
||||
'deepseek-official': {
|
||||
apiKey: 'k',
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://acme.test/v1',
|
||||
models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
// The refused swap costs a diagnostic, not the directory: every entry the
|
||||
// page needs is still declared.
|
||||
expect(ctx.llm.listConfigurableProviders()).toHaveLength(before)
|
||||
expect(ctx.llm.listConfigurableProviders().find(entry => entry.provider === 'deepseek-official')?.settingsNs)
|
||||
.toBe('llm-deepseek')
|
||||
})
|
||||
|
||||
it('replaces its entries atomically as declared routes come and go', async () => {
|
||||
const dir = await home()
|
||||
const ctx = await bootWithSettings(dir, {})
|
||||
const catalogOnly = ctx.llm.listConfigurableProviders().length
|
||||
|
||||
await ctx.settings.update(settingsNamespace('llm-pi-ai'), {
|
||||
providers: {
|
||||
'acme-gateway': {
|
||||
apiKey: 'k',
|
||||
displayName: 'Acme Gateway',
|
||||
api: 'openai-completions',
|
||||
baseURL: 'https://acme.test/v1',
|
||||
models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }],
|
||||
},
|
||||
},
|
||||
})
|
||||
expect(ctx.llm.listConfigurableProviders()).toHaveLength(catalogOnly + 1)
|
||||
expect(ctx.llm.listConfigurableProviders().find(entry => entry.provider === 'acme-gateway')?.displayName)
|
||||
.toBe('Acme Gateway')
|
||||
|
||||
await ctx.settings.replace(settingsNamespace('llm-pi-ai'), {})
|
||||
expect(ctx.llm.listConfigurableProviders()).toHaveLength(catalogOnly)
|
||||
})
|
||||
})
|
||||
24
packages/llm/llm-pi-ai/tests/config.spec.ts
Normal file
24
packages/llm/llm-pi-ai/tests/config.spec.ts
Normal file
@@ -0,0 +1,24 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { resolveProfiles } from '../src/config.ts'
|
||||
|
||||
describe('API key format', () => {
|
||||
it('trims a padded literal apiKey into the resolved profile', () => {
|
||||
const resolved = resolveProfiles({ openai: { apiKey: ' sk-abc ', baseURL: 'https://acme.test' } })
|
||||
expect(resolved.get('openai')?.apiKey).toBe('sk-abc')
|
||||
})
|
||||
|
||||
it('keeps an omitted apiKey absent so ambient authentication still applies', () => {
|
||||
const resolved = resolveProfiles({ openai: { baseURL: 'https://acme.test' } })
|
||||
expect(resolved.get('openai')?.apiKey).toBeUndefined()
|
||||
})
|
||||
|
||||
it('still tells an empty apiKey to omit itself', () => {
|
||||
expect(() => resolveProfiles({ openai: { apiKey: ' ', baseURL: 'https://acme.test' } }))
|
||||
.toThrow(/omit it to use ambient authentication/)
|
||||
})
|
||||
|
||||
it('rejects an apiKey no header can carry', () => {
|
||||
expect(() => resolveProfiles({ openai: { apiKey: 'sk-\u{1F600}', baseURL: 'https://acme.test' } }))
|
||||
.toThrow(/no HTTP header can carry/)
|
||||
})
|
||||
})
|
||||
358
packages/llm/llm-pi-ai/tests/discovery.spec.ts
Normal file
358
packages/llm/llm-pi-ai/tests/discovery.spec.ts
Normal file
@@ -0,0 +1,358 @@
|
||||
import { createServer } from 'node:http'
|
||||
import type { IncomingMessage, Server, ServerResponse } from 'node:http'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService, { userAgent } from '@deepseek-ai/dsh-llm'
|
||||
import * as LlmPiAi from '@deepseek-ai/dsh-llm-pi-ai'
|
||||
import { getBuiltinModels } from '@earendil-works/pi-ai/providers/all'
|
||||
import { discoverModels } from '../src/discovery.ts'
|
||||
|
||||
const servers: Server[] = []
|
||||
/** Credential variables a test set, cleared so the next one starts unset. */
|
||||
const touchedEnv: string[] = []
|
||||
|
||||
afterEach(async () => {
|
||||
// A no-op when the test never stubbed `fetch`; only 'probe key format'
|
||||
// below installs one.
|
||||
vi.unstubAllGlobals()
|
||||
for (const name of touchedEnv.splice(0)) Reflect.deleteProperty(process.env, name)
|
||||
await Promise.all(servers.splice(0).map(server => new Promise(resolve => server.close(resolve))))
|
||||
})
|
||||
|
||||
interface ListingServer {
|
||||
url: string
|
||||
paths: string[]
|
||||
headers: IncomingMessage['headers'][]
|
||||
}
|
||||
|
||||
/**
|
||||
* A stand-in provider that answers one scripted `GET /models`. `chunks` writes
|
||||
* without a declared length, which is how a real streamed reply arrives.
|
||||
*/
|
||||
async function listingServer(behavior: {
|
||||
status?: number
|
||||
body?: string
|
||||
chunks?: string[]
|
||||
holdOpenMs?: number
|
||||
}): Promise<ListingServer> {
|
||||
const paths: string[] = []
|
||||
const headers: IncomingMessage['headers'][] = []
|
||||
const server = createServer((request: IncomingMessage, response: ServerResponse) => {
|
||||
paths.push(request.url ?? '')
|
||||
headers.push(request.headers)
|
||||
if (behavior.chunks !== undefined) {
|
||||
// No declared length: the ceiling has to hold on what is read.
|
||||
response.writeHead(behavior.status ?? 200, { 'content-type': 'application/json' })
|
||||
for (const chunk of behavior.chunks) response.write(chunk)
|
||||
if (behavior.holdOpenMs === undefined) { response.end(); return }
|
||||
// Left open so a caller's cancellation lands while the body is still
|
||||
// being read rather than after it completed.
|
||||
setTimeout(() => { response.end() }, behavior.holdOpenMs)
|
||||
return
|
||||
}
|
||||
const body = behavior.body ?? '{}'
|
||||
response.writeHead(behavior.status ?? 200, {
|
||||
'content-type': 'application/json',
|
||||
'content-length': String(Buffer.byteLength(body)),
|
||||
})
|
||||
response.end(body)
|
||||
})
|
||||
servers.push(server)
|
||||
await new Promise<void>(resolve => server.listen(0, '127.0.0.1', resolve))
|
||||
const address = server.address()
|
||||
if (address === null || typeof address === 'string') throw new Error('no port')
|
||||
return { url: `http://127.0.0.1:${address.port}`, paths, headers }
|
||||
}
|
||||
|
||||
/** A bare dormant mount: discovery is offered whether or not a route exists. */
|
||||
async function harness(): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(LlmPiAi, {})
|
||||
return ctx
|
||||
}
|
||||
|
||||
describe('catalog-route model discovery', () => {
|
||||
it('answers from the installed registry, with capacities and no network call', async () => {
|
||||
const server = await listingServer({ body: JSON.stringify({ data: [{ id: 'from-the-endpoint' }] }) })
|
||||
const ctx = await harness()
|
||||
|
||||
const models = await ctx.llm.discoverModels('llm-pi-ai', { provider: 'deepseek', baseURL: server.url })
|
||||
|
||||
// pi-ai's own registry is the authority for its own providers, and it
|
||||
// carries what a listing endpoint would not disclose.
|
||||
expect(models.map(model => model.id).sort())
|
||||
.toEqual(getBuiltinModels('deepseek').map(model => model.id).sort())
|
||||
expect(models.every(model => (model.contextWindow ?? 0) > 0 && (model.maxTokens ?? 0) > 0)).toBe(true)
|
||||
expect(server.paths).toEqual([])
|
||||
})
|
||||
|
||||
it('needs no endpoint for a route the catalog describes', async () => {
|
||||
const ctx = await harness()
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'deepseek' })).resolves.not.toHaveLength(0)
|
||||
})
|
||||
|
||||
it('says where a route the catalog does not describe must get its models', async () => {
|
||||
const ctx = await harness()
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'acme-gateway' }))
|
||||
.rejects.toThrow(/ships no catalog for provider "acme-gateway".*set a baseURL/s)
|
||||
// A form that cleared the field says the same thing as one that never had it.
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'acme-gateway', baseURL: '' }))
|
||||
.rejects.toThrow(/set a baseURL/)
|
||||
// The seam refuses a request naming neither, so the module's own guard for
|
||||
// that shape is only reachable by calling it directly.
|
||||
await expect(discoverModels({})).rejects.toThrow(/set a baseURL/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('draft-provider model discovery', () => {
|
||||
it('reads an OpenAI-compatible listing and keeps the capacities it discloses', async () => {
|
||||
const server = await listingServer({
|
||||
body: JSON.stringify({
|
||||
data: [
|
||||
{ id: 'acme-large', display_name: 'Acme Large', context_length: 65_536, max_output_tokens: 4096 },
|
||||
{ id: 'acme-small' },
|
||||
],
|
||||
}),
|
||||
})
|
||||
const ctx = await harness()
|
||||
|
||||
const models = await ctx.llm.discoverModels('llm-pi-ai', { baseURL: `${server.url}/v1`, apiKey: 'probe-key' })
|
||||
|
||||
expect(models).toEqual([
|
||||
{ id: 'acme-large', name: 'Acme Large', contextWindow: 65_536, maxTokens: 4096 },
|
||||
{ id: 'acme-small' },
|
||||
])
|
||||
expect(server.paths).toEqual(['/v1/models'])
|
||||
expect(server.headers[0]?.authorization).toBe('Bearer probe-key')
|
||||
expect(server.headers[0]?.['user-agent']).toBe(userAgent())
|
||||
})
|
||||
|
||||
it('keeps a deployment path instead of resolving it away', async () => {
|
||||
const server = await listingServer({ body: JSON.stringify({ data: [{ id: 'm' }] }) })
|
||||
const ctx = await harness()
|
||||
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { baseURL: `${server.url}/openai/v1/` })
|
||||
|
||||
expect(server.paths).toEqual(['/openai/v1/models'])
|
||||
})
|
||||
|
||||
it('offers no credential when the draft names none', async () => {
|
||||
const server = await listingServer({ body: JSON.stringify({ data: [{ id: 'm' }] }) })
|
||||
const ctx = await harness()
|
||||
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { baseURL: server.url })
|
||||
|
||||
expect(server.headers[0]?.authorization).toBeUndefined()
|
||||
})
|
||||
|
||||
it('authenticates a configured route the draft cannot supply a key for', async () => {
|
||||
// What the Models page actually sends after a key is saved: the form holds
|
||||
// the redacted descriptor, so the draft names the route and the endpoint
|
||||
// and no credential at all. Interrogating unauthenticated would answer 401
|
||||
// and read as a wrong key.
|
||||
const server = await listingServer({ body: JSON.stringify({ data: [{ id: 'm' }] }) })
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
process.env['ACME_GATEWAY_KEY'] = 'stored-key'
|
||||
touchedEnv.push('ACME_GATEWAY_KEY')
|
||||
await ctx.plugin(LlmPiAi, {
|
||||
providers: {
|
||||
'acme-gateway': {
|
||||
apiKeyEnv: 'ACME_GATEWAY_KEY',
|
||||
api: 'openai-completions',
|
||||
baseURL: server.url,
|
||||
models: [{ id: 'acme-large' }],
|
||||
},
|
||||
},
|
||||
})
|
||||
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { provider: 'acme-gateway', baseURL: server.url })
|
||||
// A key typed into the form is the one being tested — possibly the
|
||||
// replacement for the stored one — so it wins.
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { provider: 'acme-gateway', baseURL: server.url, apiKey: 'typed' })
|
||||
// A route no profile declares yet is the create case: nothing is stored.
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { provider: 'not-declared-yet', baseURL: server.url })
|
||||
|
||||
expect(server.headers.map(headers => headers.authorization))
|
||||
.toEqual(['Bearer stored-key', 'Bearer typed', undefined])
|
||||
})
|
||||
|
||||
it('leaves a catalog route\'s credential unresolved, having never reached the network', async () => {
|
||||
// The catalog answers before any endpoint is asked, so a route whose
|
||||
// profile names a credential that is not set must still answer rather than
|
||||
// failing over a key the interrogation never needed.
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
Reflect.deleteProperty(process.env, 'ABSENT_FOR_DISCOVERY')
|
||||
await ctx.plugin(LlmPiAi, { providers: { deepseek: { apiKeyEnv: 'ABSENT_FOR_DISCOVERY' } } })
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'deepseek' })).resolves.not.toHaveLength(0)
|
||||
})
|
||||
|
||||
it('drops unusable rows rather than failing the whole listing', async () => {
|
||||
const server = await listingServer({
|
||||
body: JSON.stringify({
|
||||
data: [
|
||||
{ id: 'good' },
|
||||
{ id: '' },
|
||||
{ name: 'no id at all' },
|
||||
null,
|
||||
{ id: 'good' },
|
||||
{ id: 'zero-capacity', context_length: 0, max_tokens: -1 },
|
||||
],
|
||||
}),
|
||||
})
|
||||
const ctx = await harness()
|
||||
|
||||
expect(await ctx.llm.discoverModels('llm-pi-ai', { baseURL: server.url }))
|
||||
.toEqual([{ id: 'good' }, { id: 'zero-capacity' }])
|
||||
})
|
||||
|
||||
it('points at the credential for a rejected one, and only then', async () => {
|
||||
const ctx = await harness()
|
||||
|
||||
for (const status of [401, 403]) {
|
||||
const refused = await listingServer({ status, body: '{"error":"nope"}' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: refused.url, apiKey: 'wrong' }))
|
||||
.rejects.toThrow(new RegExp(`answered ${status}; check the API key`))
|
||||
}
|
||||
|
||||
// A server fault is not a credential problem, so it must not send the user
|
||||
// off to re-check a key that is fine.
|
||||
const broken = await listingServer({ status: 500, body: '{"error":"boom"}' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: broken.url, apiKey: 'fine' }))
|
||||
.rejects.toThrow(/answered 500$/)
|
||||
})
|
||||
|
||||
it('reports a reply that is not a model listing', async () => {
|
||||
const server = await listingServer({ body: '{"models":[]}' })
|
||||
const ctx = await harness()
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: server.url }))
|
||||
.rejects.toThrow(/no "data" array; enter this provider's models by hand/)
|
||||
|
||||
const broken = await listingServer({ body: 'not json at all' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: broken.url }))
|
||||
.rejects.toThrow(/did not answer with JSON/)
|
||||
})
|
||||
|
||||
it('refuses an oversized reply, whether its length is declared or streamed', async () => {
|
||||
const ctx = await harness()
|
||||
// Just over the four-megabyte ceiling, as one padded model row.
|
||||
const oversized = `{"data":[{"id":"m","pad":"${'x'.repeat(4 * 1024 * 1024)}"}]}`
|
||||
|
||||
const declared = await listingServer({ body: oversized })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: declared.url }))
|
||||
.rejects.toThrow(/answered with more than 4194304 bytes/)
|
||||
|
||||
// A streamed reply declares no length, so the ceiling has to hold on the
|
||||
// body the harness actually read.
|
||||
const streamed = await listingServer({ chunks: ['{"data":[{"id":"m","pad":"', 'x'.repeat(4 * 1024 * 1024), '"}]}'] })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: streamed.url }))
|
||||
.rejects.toThrow(/answered with more than 4194304 bytes/)
|
||||
})
|
||||
|
||||
it('reports an unreachable endpoint instead of an empty catalog', async () => {
|
||||
const ctx = await harness()
|
||||
// Port 9 is the discard service: nothing accepts a connection there.
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: 'http://127.0.0.1:9/v1' }))
|
||||
.rejects.toMatchObject({ code: 'DISCOVERY_FAILED' })
|
||||
})
|
||||
|
||||
it.each(['anthropic-messages', 'azure-openai-responses', 'openai-codex-responses', 'google-generative-ai'])(
|
||||
'says it cannot interrogate %s rather than guessing a shape',
|
||||
async (api) => {
|
||||
// Azure authenticates with an `api-key` header and an `api-version`
|
||||
// query despite its OpenAI lineage, and Codex uses OAuth; guessing at
|
||||
// either would report an auth failure as a provider with no models.
|
||||
const ctx = await harness()
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: 'https://gateway.example/v1', api }))
|
||||
.rejects.toMatchObject({ code: 'DISCOVERY_UNSUPPORTED' })
|
||||
},
|
||||
)
|
||||
|
||||
it('reports cancellation during the body read as an abort, not a raw reason', async () => {
|
||||
const ctx = await harness()
|
||||
const controller = new AbortController()
|
||||
// Chunked, so the headers arrive and the cancellation lands mid-body.
|
||||
const slow = await listingServer({ chunks: ['{"data":[', '{"id":"a"}'], holdOpenMs: 400 })
|
||||
const probe = ctx.llm.discoverModels('llm-pi-ai', { baseURL: slow.url, signal: controller.signal })
|
||||
setTimeout(() => { controller.abort('test cancellation') }, 40)
|
||||
|
||||
await expect(probe).rejects.toMatchObject({ code: 'ABORTED' })
|
||||
})
|
||||
|
||||
it('honors caller cancellation', async () => {
|
||||
const ctx = await harness()
|
||||
const aborted = AbortSignal.abort('test cancellation')
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', {
|
||||
baseURL: 'http://127.0.0.1:9/v1',
|
||||
signal: aborted,
|
||||
})).rejects.toMatchObject({ code: 'ABORTED' })
|
||||
})
|
||||
|
||||
it('is offered for the namespace, and refuses one it does not serve', async () => {
|
||||
const ctx = await harness()
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'openai' })).resolves.not.toHaveLength(0)
|
||||
await expect(ctx.llm.discoverModels('llm-deepseek', { baseURL: 'https://api.deepseek.com' }))
|
||||
.rejects.toMatchObject({ code: 'NO_DISCOVERY' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: '' }))
|
||||
.rejects.toMatchObject({ code: 'INVALID_DISCOVERY' })
|
||||
})
|
||||
|
||||
it('withdraws the offer when the plugin unloads', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const fiber = await ctx.plugin(LlmPiAi, {})
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'openai' })).resolves.not.toHaveLength(0)
|
||||
|
||||
await fiber.dispose()
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { provider: 'openai' }))
|
||||
.rejects.toMatchObject({ code: 'NO_DISCOVERY' })
|
||||
})
|
||||
})
|
||||
|
||||
describe('probe key format', () => {
|
||||
it('reports an illegal probe key as a credential fault, not an unreachable endpoint', async () => {
|
||||
await expect(discoverModels({
|
||||
baseURL: 'https://acme.test',
|
||||
api: 'openai-completions',
|
||||
apiKey: 'sk-\u{1F600}',
|
||||
})).rejects.toMatchObject({ code: 'INVALID_CREDENTIAL' })
|
||||
})
|
||||
|
||||
it('reports a blank probe key as a credential fault too', async () => {
|
||||
// The Models page omits `apiKey` entirely for a cleared field rather than
|
||||
// sending '', so this pins the contract for every other caller: a supplied
|
||||
// key is judged, and only an absent one probes unauthenticated. '' means
|
||||
// "I have a key" and is answered as the empty key it is.
|
||||
await expect(discoverModels({
|
||||
baseURL: 'https://acme.test',
|
||||
api: 'openai-completions',
|
||||
apiKey: '',
|
||||
})).rejects.toMatchObject({ code: 'INVALID_CREDENTIAL' })
|
||||
})
|
||||
|
||||
it('leaves a probe with no key unauthenticated', async () => {
|
||||
// The file's other cases capture headers through a real local HTTP server
|
||||
// (`listingServer`); this one has no route or stored key to resolve, so
|
||||
// the smallest real double is a `fetch` stub, scoped to this test and
|
||||
// unstubbed by the shared `afterEach` above.
|
||||
const requests: RequestInit[] = []
|
||||
vi.stubGlobal('fetch', async (_url: string | URL, init?: RequestInit) => {
|
||||
requests.push(init ?? {})
|
||||
return new Response(JSON.stringify({ data: [] }), {
|
||||
status: 200,
|
||||
headers: { 'content-type': 'application/json' },
|
||||
})
|
||||
})
|
||||
|
||||
await discoverModels({ baseURL: 'https://acme.test', api: 'openai-completions' })
|
||||
|
||||
const headers = new Headers(requests[0]?.headers)
|
||||
expect(headers.has('authorization')).toBe(false)
|
||||
})
|
||||
})
|
||||
@@ -68,6 +68,7 @@ describe('request-level dynamic profiles', () => {
|
||||
displayName: 'openai',
|
||||
settingsNs: 'llm-pi-ai',
|
||||
settingsPath: ['providers', 'openai'],
|
||||
declared: false,
|
||||
})
|
||||
await ctx.settings.update(NS, {
|
||||
providers: { deepseek: { apiKeyEnv: 'PI_DYNAMIC_KEY', baseURL: server.url } },
|
||||
@@ -105,8 +106,8 @@ describe('request-level dynamic profiles', () => {
|
||||
// composition route stays.
|
||||
await ctx.settings.replace(NS, {})
|
||||
expect(ctx.llm.listProviders().map(provider => provider.id)).toEqual(['openai'])
|
||||
await expect(assemble(ctx, { provider: 'deepseek', model: 'deepseek-v4-flash', messages: [] }))
|
||||
.rejects.toMatchObject({ code: 'NO_ADAPTER' })
|
||||
const removed = await assemble(ctx, { provider: 'deepseek', model: 'deepseek-v4-flash', messages: [] })
|
||||
expect(removed.finish).toMatchObject({ kind: 'error', failure: { code: 'NO_ADAPTER' } })
|
||||
})
|
||||
|
||||
it('rotates the per-request credential referenced by apiKeyEnv', async () => {
|
||||
@@ -146,13 +147,16 @@ describe('request-level dynamic profiles', () => {
|
||||
expect(ctx.llm.listProviders().map(provider => provider.id)).toEqual(['openai'])
|
||||
})
|
||||
|
||||
it('keeps the last good profiles when a settings snapshot names an unknown provider', async () => {
|
||||
it('refuses a settings write this adapter could not serve, leaving its routes alone', async () => {
|
||||
const dir = await home()
|
||||
const ctx = await boot(dir, { providers: { openai: {} } })
|
||||
|
||||
// Schema-valid but catalog-invalid: the resolver rejects it and the
|
||||
// last good route set keeps serving.
|
||||
await ctx.settings.update(NS, { providers: { 'not-a-real-provider': {} } })
|
||||
// Shape-valid but unserviceable: a route the catalog does not ship and
|
||||
// that lists no models of its own. The section schema resolves the whole
|
||||
// profile set, so this is refused where it is written rather than stored
|
||||
// and then quietly disabling every route in the namespace.
|
||||
await expect(ctx.settings.update(NS, { providers: { 'not-a-real-provider': {} } }))
|
||||
.rejects.toThrow(/resolves no models/)
|
||||
expect(ctx.llm.listProviders().map(provider => provider.id)).toEqual(['openai'])
|
||||
})
|
||||
|
||||
|
||||
@@ -1,41 +1,74 @@
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
import type { StreamChunk } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
const streamSimple = vi.hoisted(() => vi.fn())
|
||||
|
||||
// The 0.81 SDK moved `streamSimple` to the compat entry; the adapter imports it
|
||||
// from there, so the mock must target the same specifier.
|
||||
vi.mock('@earendil-works/pi-ai/compat', async (importOriginal) => {
|
||||
const actual = await importOriginal<typeof import('@earendil-works/pi-ai/compat')>()
|
||||
return { ...actual, streamSimple }
|
||||
})
|
||||
// A hand-declared route is built by `createProvider` over the protocol table in
|
||||
// `src/provider.ts`, so the table's lazy api module is the SDK boundary this
|
||||
// test can observe. A catalog route dispatches through pi-ai's own provider and
|
||||
// would not see this mock.
|
||||
vi.mock('@earendil-works/pi-ai/api/openai-completions.lazy', () => ({
|
||||
openAICompletionsApi: () => ({ stream: streamSimple, streamSimple }),
|
||||
}))
|
||||
|
||||
import { PiAiAdapter } from '../src/adapter.ts'
|
||||
import { resolveProfiles } from '../src/config.ts'
|
||||
|
||||
afterEach(() => { streamSimple.mockReset() })
|
||||
|
||||
/** A hand-declared OpenAI-compatible route with one fully described model. */
|
||||
function gatewayAdapter(): PiAiAdapter {
|
||||
return new PiAiAdapter({
|
||||
profiles: () => resolveProfiles({
|
||||
'local-gateway': {
|
||||
apiKey: 'test-key',
|
||||
api: 'openai-completions',
|
||||
baseURL: 'http://127.0.0.1:9/v1',
|
||||
models: [{ id: 'local-model', contextWindow: 8192, maxTokens: 1024 }],
|
||||
},
|
||||
}),
|
||||
resolveApiKey: () => Promise.resolve('test-key'),
|
||||
})
|
||||
}
|
||||
|
||||
async function drain(adapter: PiAiAdapter): Promise<StreamChunk[]> {
|
||||
const chunks: StreamChunk[] = []
|
||||
for await (const chunk of adapter.stream({
|
||||
provider: 'local-gateway',
|
||||
model: 'local-model',
|
||||
messages: [],
|
||||
})) chunks.push(chunk)
|
||||
return chunks
|
||||
}
|
||||
|
||||
describe('pi-ai SDK retry boundary', () => {
|
||||
it('pins one SDK attempt even when the installed provider currently defaults to zero retries', async () => {
|
||||
const failure = new Error('mock SDK boundary')
|
||||
streamSimple.mockReturnValue({
|
||||
async * [Symbol.asyncIterator](): AsyncGenerator<never> {
|
||||
throw failure
|
||||
},
|
||||
})
|
||||
const adapter = new PiAiAdapter({
|
||||
profiles: () => resolveProfiles({ openai: { apiKey: 'test-key' } }),
|
||||
resolveApiKey: () => Promise.resolve('test-key'),
|
||||
})
|
||||
const drain = async (): Promise<void> => {
|
||||
for await (const _chunk of adapter.stream({
|
||||
provider: 'openai',
|
||||
model: 'gpt-4.1',
|
||||
messages: [],
|
||||
})) { /* drain */ }
|
||||
}
|
||||
streamSimple.mockImplementation(() => { throw new Error('mock SDK boundary') })
|
||||
|
||||
const chunks = await drain(gatewayAdapter())
|
||||
|
||||
await expect(drain()).rejects.toBe(failure)
|
||||
expect(streamSimple).toHaveBeenCalledOnce()
|
||||
expect(streamSimple.mock.calls[0]?.[2]).toMatchObject({ maxRetries: 0 })
|
||||
expect(streamSimple.mock.calls[0]?.[2]).toMatchObject({ maxRetries: 0, apiKey: 'test-key' })
|
||||
// pi-ai reports a setup failure as a terminal in-stream error rather than
|
||||
// throwing, which the converter turns into the harness error finish.
|
||||
expect(chunks.at(-1)).toMatchObject({
|
||||
type: 'finish',
|
||||
reason: { kind: 'error', failure: { message: 'mock SDK boundary' } },
|
||||
})
|
||||
})
|
||||
|
||||
it('dispatches a hand-declared route to the endpoint and model its configuration describes', async () => {
|
||||
streamSimple.mockImplementation(() => { throw new Error('mock SDK boundary') })
|
||||
|
||||
await drain(gatewayAdapter())
|
||||
|
||||
expect(streamSimple.mock.calls[0]?.[0]).toMatchObject({
|
||||
id: 'local-model',
|
||||
provider: 'local-gateway',
|
||||
api: 'openai-completions',
|
||||
baseUrl: 'http://127.0.0.1:9/v1',
|
||||
contextWindow: 8192,
|
||||
maxTokens: 1024,
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/llm-retry/README.md
|
||||
README.md: 8de3ea8c9321f04f5af1b0d7ab361f73eaabc822
|
||||
README.zh.md: 978854e9466e271535a10fcea406d0dcb5607285
|
||||
README.md: 23b55a30989cc51d4dd9076b61b6595452b0abd0
|
||||
README.zh.md: 267ef12a87561fd8effef726a781e505225baf03
|
||||
|
||||
@@ -48,6 +48,6 @@ The reconstructed request preserves the prior prefix and is eligible for provide
|
||||
|
||||
- **Agent turns are the only retry boundary** — direct `ctx.llm.stream()` consumers remain single-attempt because a raw stream cannot separate already-emitted chunks durably.
|
||||
- **Always mode retries permanent failures** — authentication, quota, invalid-request, protocol, and unrecoverable context errors continue until success, cancellation, or disposal; deployments own provider-specific cost and latency controls.
|
||||
- **Finite plugin budgets add** — normal mode counts only its configured codes and exact provider policy, while context-overflow compaction owns a separate budget. A future overlapping policy must document and test registration-order behavior.
|
||||
- **Finite plugin budgets add** — normal mode counts only its configured codes and exact provider policy, while context-overflow compaction owns a separate budget. Any overlapping policy must define registration-order behavior.
|
||||
- **Recovery policies compose by waterfall order** — always mode accepts a downstream retry before applying its fallback. A later policy that ignores cancellation and never settles also prevents fallback, turn quiescence, and plugin disposal from completing.
|
||||
- **`llm/retry` records scheduling, not completion** — later step and turn events establish success, exhaustion, or cancellation.
|
||||
|
||||
@@ -48,6 +48,6 @@
|
||||
|
||||
- **agent 轮次是唯一重试边界**:直接 `ctx.llm.stream()` 消费方仍只尝试一次,因为原始流无法持久地区分各次尝试已经发出的分片。
|
||||
- **always mode 会重试永久性失败**:身份验证、配额、无效请求、协议和无法恢复的上下文错误都会继续重试,直至成功、取消或 dispose;部署负责提供方特定的成本与延迟控制。
|
||||
- **有限插件预算可叠加**:normal mode 只统计已配置 code 和确切提供方策略,上下文溢出压缩(compaction)则拥有独立预算。未来如有重叠策略,必须记录并测试注册顺序行为。
|
||||
- **有限插件预算可叠加**:normal mode 只统计已配置 code 和确切提供方策略,上下文溢出压缩(compaction)则拥有独立预算。任何重叠策略都必须定义注册顺序行为。
|
||||
- **恢复策略按 waterfall 顺序组合**:always mode 会先接受下游重试,再应用自己的回退。后续策略如果忽略取消且永不结算,也会阻止回退、轮次完全停稳和插件 dispose 完成。
|
||||
- **`llm/retry` 记录调度,不是完成**:后续步骤与轮次事件用于确立成功、耗尽或取消。
|
||||
|
||||
@@ -25,9 +25,7 @@
|
||||
"lib/index.js",
|
||||
"lib/invariant.js",
|
||||
"lib/types/**/*.js",
|
||||
"lib/types/**/*.d.ts",
|
||||
"lib/types/**/*.d.ts.map",
|
||||
"src"
|
||||
"lib/types/**/*.d.ts"
|
||||
],
|
||||
"license": "BSD-3-Clause",
|
||||
"peerDependencies": {
|
||||
|
||||
@@ -1,28 +1,29 @@
|
||||
/** Durable request-route lookup for one closed model step. @module @deepseek-ai/dsh-llm-retry/history */
|
||||
/** Durable request-route lookup for one open model step. @module @deepseek-ai/dsh-llm-retry/history */
|
||||
|
||||
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
|
||||
/**
|
||||
* Find the provider in force when one step closed, excluding later recovery mutations.
|
||||
* Find the provider in force for one currently open step.
|
||||
* Request headers remain effective across turn boundaries until a newer full
|
||||
* snapshot changes them; every provider change requires a newer full snapshot.
|
||||
* @param events - session events containing the closed step.
|
||||
* @param events - session events ending inside the open step.
|
||||
* @param turn - turn that owns the failed step.
|
||||
* @param step - failed step whose provider is required.
|
||||
* @returns the provider from the request header in force at that step boundary.
|
||||
* @returns the provider from the request header in force for the step.
|
||||
*/
|
||||
export function providerForClosedStep(
|
||||
export function providerForOpenStep(
|
||||
events: readonly SessionEvent[],
|
||||
turn: number,
|
||||
step: number,
|
||||
): string | undefined {
|
||||
const stepEndIndex = events.findLastIndex(event =>
|
||||
event.type === 'step/end'
|
||||
const stepStartIndex = events.findLastIndex(event =>
|
||||
event.type === 'step/start'
|
||||
&& event.data.turn === turn
|
||||
&& event.data.step === step,
|
||||
)
|
||||
if (stepEndIndex < 0) return undefined
|
||||
for (let index = stepEndIndex; index >= 0; index -= 1) {
|
||||
if (stepStartIndex < 0 || events.slice(stepStartIndex + 1).some(event =>
|
||||
event.type === 'step/end' || event.type === 'turn/end')) return undefined
|
||||
for (let index = events.length - 1; index >= 0; index -= 1) {
|
||||
// The loop bounds prove this indexed read exists.
|
||||
// oxlint-disable-next-line typescript/no-non-null-assertion
|
||||
const event = events[index]!
|
||||
|
||||
@@ -1,20 +1,19 @@
|
||||
/**
|
||||
* Provider-routed model-request retry policy on the agent loop's closed-step
|
||||
* Provider-routed model-request retry policy on the agent loop's request
|
||||
* recovery seam. Each scheduled retry is durable before its cancellable wait.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-llm-retry
|
||||
*/
|
||||
|
||||
import type { Context } from 'cordis'
|
||||
import type { Context, Events } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import type { Agent, RequestError, RequestErrorAction } from '@deepseek-ai/dsh-agent'
|
||||
import type { Agent, RequestErrorAction } from '@deepseek-ai/dsh-agent'
|
||||
import type { LlmFailure, ResolvedRetryPolicy } from '@deepseek-ai/dsh-llm'
|
||||
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import { providerForClosedStep } from './history.ts'
|
||||
|
||||
declare module '@deepseek-ai/dsh-session' {
|
||||
interface SessionEventMap {
|
||||
/** Durable, non-surface record of one provider-routed retry scheduled after a closed failed step. */
|
||||
/** Durable, non-surface record of one provider-routed retry scheduled after a failed request attempt. */
|
||||
'llm/retry': {
|
||||
turn: number
|
||||
step: number
|
||||
@@ -173,25 +172,10 @@ export function apply(ctx: Context, config: Config = {}, internals: RetryInterna
|
||||
}
|
||||
|
||||
async function recover(
|
||||
agent: Agent,
|
||||
turn: number,
|
||||
step: number,
|
||||
_error: RequestError,
|
||||
failure: LlmFailure,
|
||||
priorFailures: readonly LlmFailure[],
|
||||
policy: ResolvedRetryPolicy | undefined,
|
||||
signal: AbortSignal,
|
||||
{ agent, turn, step, provider, failure, retryPolicy: policy, signal }: Parameters<Events['agent/request-error']>[0],
|
||||
next: () => Promise<RequestErrorAction>,
|
||||
): Promise<RequestErrorAction> {
|
||||
if (policy === undefined) return next()
|
||||
// The call-local policy belongs to the registration that served this
|
||||
// failure. Recover only the durable provider identity from the header;
|
||||
// downstream recovery may append later state before an always fallback.
|
||||
const provider = providerForClosedStep(agent.session.events, turn, step)
|
||||
/* v8 ignore next 3 -- agent-loop closes only steps whose request header was recorded */
|
||||
if (provider === undefined) {
|
||||
throw new Error(`llm-retry: no request provider for closed turn ${turn}/step ${step}`)
|
||||
}
|
||||
if (policy.mode === 'always') {
|
||||
if (signal.aborted || lifetime.signal.aborted) return
|
||||
const fusedSignal = AbortSignal.any([signal, lifetime.signal])
|
||||
@@ -213,11 +197,10 @@ export function apply(ctx: Context, config: Config = {}, internals: RetryInterna
|
||||
}
|
||||
|
||||
const policyKey = retryPolicyKey(policy)
|
||||
const firstPriorTurn = turn - priorFailures.length
|
||||
const priorPolicyRetry = agent.session.events.findLast((event): event is SessionEvent<'llm/retry'> =>
|
||||
event.type === 'llm/retry'
|
||||
&& event.data.turn >= firstPriorTurn
|
||||
&& event.data.turn < turn
|
||||
&& event.data.turn === turn
|
||||
&& event.data.step === step
|
||||
&& event.data.provider === provider
|
||||
&& event.data.policyKey === policyKey,
|
||||
)
|
||||
@@ -242,21 +225,14 @@ export function apply(ctx: Context, config: Config = {}, internals: RetryInterna
|
||||
}
|
||||
|
||||
const disposeListener = ctx.on('agent/request-error', (
|
||||
agent: Agent,
|
||||
turn: number,
|
||||
step: number,
|
||||
error: RequestError,
|
||||
failure: LlmFailure,
|
||||
priorFailures: readonly LlmFailure[],
|
||||
policy: ResolvedRetryPolicy | undefined,
|
||||
signal: AbortSignal,
|
||||
payload,
|
||||
next: () => Promise<RequestErrorAction>,
|
||||
) => {
|
||||
// A waterfall may have captured this callback before its registration was
|
||||
// removed. Lifetime cancellation must prevent that stale callback from
|
||||
// entering a downstream policy after disposal.
|
||||
if (lifetime.signal.aborted) return Promise.resolve<RequestErrorAction>(undefined)
|
||||
return track(recover(agent, turn, step, error, failure, priorFailures, policy, signal, next))
|
||||
return track(recover(payload, next))
|
||||
})
|
||||
|
||||
ctx.effect(() => async () => {
|
||||
|
||||
@@ -5,7 +5,7 @@ import type { Session, SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import type { LlmFailure } from '@deepseek-ai/dsh-llm'
|
||||
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
|
||||
import type { InvariantFailure, InvariantInstaller } from '@deepseek-ai/dsh-invariants'
|
||||
import { providerForClosedStep } from './history.ts'
|
||||
import { providerForOpenStep } from './history.ts'
|
||||
import type {} from './index.ts'
|
||||
|
||||
const PACKAGE_NAME = '@deepseek-ai/dsh-llm-retry'
|
||||
@@ -41,35 +41,7 @@ function validateFailure(value: unknown, fail: InvariantFailure): asserts value
|
||||
}
|
||||
}
|
||||
|
||||
/** Find the first turn in the structured-failure retry chain containing `turn`. */
|
||||
function retryChainStart(history: readonly SessionEvent[], turn: number): number {
|
||||
let startIndex = history.findLastIndex(
|
||||
event => event.type === 'turn/start' && event.data.turn === turn,
|
||||
)
|
||||
while (startIndex >= 0) {
|
||||
const start = history[startIndex]
|
||||
if (start?.type !== 'turn/start' || start.data.trigger.kind !== 'retry') break
|
||||
|
||||
let endIndex = startIndex - 1
|
||||
while (endIndex >= 0 && history[endIndex]?.type !== 'turn/end') endIndex -= 1
|
||||
const end = history[endIndex]
|
||||
if (end?.type !== 'turn/end'
|
||||
|| end.data.reason.kind !== 'error'
|
||||
|| end.data.reason.failure === undefined) break
|
||||
|
||||
const previousStart = history.findLastIndex(
|
||||
(event, index) =>
|
||||
index < endIndex
|
||||
&& event.type === 'turn/start'
|
||||
&& event.data.turn === end.data.turn,
|
||||
)
|
||||
if (previousStart < 0) break
|
||||
startIndex = previousStart
|
||||
}
|
||||
return startIndex
|
||||
}
|
||||
|
||||
/** Validate one retry record against the open turn and most recently closed step. */
|
||||
/** Validate one retry record against the currently open request step. */
|
||||
function validateRetry(
|
||||
history: readonly SessionEvent[],
|
||||
event: SessionEvent<'llm/retry'>,
|
||||
@@ -106,49 +78,34 @@ function validateRetry(
|
||||
fail(`llm/retry delayMs must be a finite number within 0..${MAX_TIMER_DELAY_MS}`)
|
||||
}
|
||||
|
||||
const currentTurnEvents: SessionEvent[] = []
|
||||
let openTurn: number | undefined
|
||||
for (const prior of history.slice().reverse()) {
|
||||
if (prior.type === 'turn/end') fail('llm/retry must be appended inside an open turn')
|
||||
if (prior.type === 'turn/start') {
|
||||
openTurn = prior.data.turn
|
||||
break
|
||||
}
|
||||
currentTurnEvents.push(prior)
|
||||
const turnBoundary = history.findLast(prior =>
|
||||
prior.type === 'turn/start' || prior.type === 'turn/end')
|
||||
if (turnBoundary?.type !== 'turn/start') {
|
||||
fail('llm/retry must be appended inside an open turn')
|
||||
}
|
||||
if (openTurn === undefined) fail('llm/retry must be appended inside an open turn')
|
||||
if (turn !== openTurn) {
|
||||
fail(`llm/retry names turn ${turn}, but the open turn is ${openTurn}`)
|
||||
if (turn !== turnBoundary.data.turn) {
|
||||
fail(`llm/retry names turn ${turn}, but the open turn is ${turnBoundary.data.turn}`)
|
||||
}
|
||||
|
||||
let closedStep: number | undefined
|
||||
for (const prior of currentTurnEvents) {
|
||||
if (prior.type === 'step/start') {
|
||||
fail(`llm/retry must follow step/end, but step ${prior.data.step} is still open`)
|
||||
}
|
||||
if (prior.type === 'step/end') {
|
||||
closedStep = prior.data.step
|
||||
break
|
||||
}
|
||||
const stepBoundary = history.findLast(prior =>
|
||||
prior.type === 'step/start' || prior.type === 'step/end')
|
||||
if (stepBoundary?.type !== 'step/start') {
|
||||
fail('llm/retry must be appended inside an open step')
|
||||
}
|
||||
if (closedStep === undefined || step !== closedStep) {
|
||||
fail(`llm/retry names step ${step}, but the latest closed step is ${String(closedStep)}`)
|
||||
if (step !== stepBoundary.data.step || turn !== stepBoundary.data.turn) {
|
||||
fail(`llm/retry names turn ${turn}/step ${step}, but the open step is ${stepBoundary.data.turn}/${stepBoundary.data.step}`)
|
||||
}
|
||||
const routedProvider = providerForClosedStep(history, turn, step)
|
||||
const routedProvider = providerForOpenStep(history, turn, step)
|
||||
if (routedProvider !== provider) {
|
||||
fail(`llm/retry provider ${provider} does not match the failed request provider ${String(routedProvider)}`)
|
||||
}
|
||||
|
||||
const chainStart = retryChainStart(history, turn)
|
||||
const chain = history.slice(Math.max(chainStart, 0))
|
||||
const lastSuccess = chain.findLastIndex(prior => prior.type === 'assistant/message')
|
||||
const chainRetries = chain.slice(lastSuccess + 1)
|
||||
.filter((prior): prior is SessionEvent<'llm/retry'> => prior.type === 'llm/retry')
|
||||
if (chainRetries.some(prior => prior.data.turn === turn && prior.data.step === step)) {
|
||||
fail(`llm/retry duplicates the retry record for turn ${turn}/step ${step}`)
|
||||
}
|
||||
const priorPolicyRetry = chainRetries.findLast(prior =>
|
||||
prior.data.provider === provider && prior.data.policyKey === policyKey)
|
||||
const priorPolicyRetry = history.findLast((prior): prior is SessionEvent<'llm/retry'> =>
|
||||
prior.type === 'llm/retry'
|
||||
&& prior.data.turn === turn
|
||||
&& prior.data.step === step
|
||||
&& prior.data.provider === provider
|
||||
&& prior.data.policyKey === policyKey)
|
||||
const expectedRetry = (priorPolicyRetry?.data.retry ?? 0) + 1
|
||||
if (retry !== expectedRetry) {
|
||||
fail(`llm/retry retry ${retry} must equal provider policy retry ${expectedRetry}`)
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import SessionStore, { SessionId, type Session } from '@deepseek-ai/dsh-session'
|
||||
import { createUserMessage, ProviderRequestId , createMessage } from '@deepseek-ai/dsh-llm'
|
||||
import { createUserMessage, ProviderRequestId } from '@deepseek-ai/dsh-llm'
|
||||
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
|
||||
import InvariantService from '@deepseek-ai/dsh-invariants'
|
||||
import * as RetryInvariant from '@deepseek-ai/dsh-llm-retry/invariant'
|
||||
import { providerForClosedStep } from '../src/history.ts'
|
||||
import { providerForOpenStep } from '../src/history.ts'
|
||||
|
||||
async function setup(): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
@@ -15,26 +15,24 @@ async function setup(): Promise<Context> {
|
||||
return ctx
|
||||
}
|
||||
|
||||
function closeStep(ctx: Context, id: string, turn = 1, step = 1) {
|
||||
function openStep(ctx: Context, id: string, turn = 1, step = 1) {
|
||||
const session = ctx.sessions.create(SessionId(id))
|
||||
session.append('turn/start', { turn, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('turn/start', { turn })
|
||||
session.append('step/start', { turn, step })
|
||||
session.append('request/header', {
|
||||
header: { config: { provider: 'mock', model: 'mock' } },
|
||||
reason: 'initial',
|
||||
})
|
||||
session.append('step/end', { turn, step })
|
||||
return session
|
||||
}
|
||||
|
||||
function appendRetryTurn(session: Session, turn: number) {
|
||||
session.append('turn/start', { turn, trigger: { kind: 'retry' } })
|
||||
session.append('turn/start', { turn })
|
||||
session.append('step/start', { turn, step: 1 })
|
||||
session.append('request/header', {
|
||||
header: { config: { provider: 'mock', model: 'mock' } },
|
||||
reason: 'initial',
|
||||
})
|
||||
session.append('step/end', { turn, step: 1 })
|
||||
session.append('llm/retry', { turn, step: 1, ...normal })
|
||||
}
|
||||
|
||||
@@ -58,28 +56,24 @@ const always = {
|
||||
}
|
||||
|
||||
describe('llm-retry invariants', () => {
|
||||
it('has no provider without the requested closed step or a route marker', () => {
|
||||
expect(providerForClosedStep([], 1, 1)).toBeUndefined()
|
||||
expect(providerForClosedStep([{
|
||||
type: 'step/end',
|
||||
it('has no provider without the requested open step or a route marker', () => {
|
||||
expect(providerForOpenStep([], 1, 1)).toBeUndefined()
|
||||
expect(providerForOpenStep([{
|
||||
type: 'step/start',
|
||||
data: { turn: 1, step: 1 },
|
||||
}] as never, 1, 1)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('accepts bounded and unbounded records after successive closed steps', async () => {
|
||||
it('accepts successive bounded and unbounded records inside their open steps', async () => {
|
||||
const ctx = await setup()
|
||||
const session = closeStep(ctx, 'retry-invariant-valid')
|
||||
const session = openStep(ctx, 'retry-invariant-valid')
|
||||
|
||||
expect(() => {
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
session.append('turn/end', { turn: 1, reason: { kind: 'error', step: 1, failure } })
|
||||
session.append('turn/start', { turn: 2, trigger: { kind: 'retry' } })
|
||||
session.append('step/start', { turn: 2, step: 1 })
|
||||
session.append('step/end', { turn: 2, step: 1 })
|
||||
session.append('llm/retry', {
|
||||
turn: 2, step: 1, ...normal, retry: 2, delayMs: 0,
|
||||
turn: 1, step: 1, ...normal, retry: 2, delayMs: 0,
|
||||
})
|
||||
const unbounded = closeStep(ctx, 'retry-invariant-always')
|
||||
const unbounded = openStep(ctx, 'retry-invariant-always')
|
||||
unbounded.append('llm/retry', { turn: 1, step: 1, ...always })
|
||||
}).not.toThrow()
|
||||
expect(() => { ctx.emit('tools/change') }).not.toThrow()
|
||||
@@ -87,7 +81,7 @@ describe('llm-retry invariants', () => {
|
||||
|
||||
it('validates the complete durable failure payload', async () => {
|
||||
const ctx = await setup()
|
||||
const complete = closeStep(ctx, 'retry-invariant-complete-failure')
|
||||
const complete = openStep(ctx, 'retry-invariant-complete-failure')
|
||||
expect(() => {
|
||||
complete.append('llm/retry', {
|
||||
turn: 1,
|
||||
@@ -126,7 +120,7 @@ describe('llm-retry invariants', () => {
|
||||
['request-id-empty', { message: 'failed', code: 'RATE_LIMIT', requestId: '' }, /failure\.requestId/],
|
||||
]
|
||||
for (const [name, invalidFailure, message] of invalidFailures) {
|
||||
const session = closeStep(ctx, `retry-invariant-failure-${name}`)
|
||||
const session = openStep(ctx, `retry-invariant-failure-${name}`)
|
||||
expect(() => {
|
||||
session.append('llm/retry', {
|
||||
turn: 1, step: 1, ...always, failure: invalidFailure,
|
||||
@@ -150,95 +144,75 @@ describe('llm-retry invariants', () => {
|
||||
['delay-type', { ...normal, delayMs: '1' }, /delayMs/],
|
||||
])('rejects invalid retry data: %s', async (name, data, message) => {
|
||||
const ctx = await setup()
|
||||
const session = closeStep(ctx, `retry-invariant-${name}`)
|
||||
const session = openStep(ctx, `retry-invariant-${name}`)
|
||||
expect(() => {
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...data } as never)
|
||||
}).toThrow(message)
|
||||
})
|
||||
|
||||
it('rejects records outside the latest closed step of an open turn', async () => {
|
||||
it('rejects records outside the currently open turn and step', async () => {
|
||||
const ctx = await setup()
|
||||
const absent = ctx.sessions.create(SessionId('retry-invariant-no-turn'))
|
||||
expect(() => {
|
||||
absent.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
}).toThrow(/inside an open turn/)
|
||||
|
||||
const wrongTurn = closeStep(ctx, 'retry-invariant-wrong-turn')
|
||||
const wrongTurn = openStep(ctx, 'retry-invariant-wrong-turn')
|
||||
expect(() => {
|
||||
wrongTurn.append('llm/retry', { turn: 2, step: 1, ...normal })
|
||||
}).toThrow(/open turn is 1/)
|
||||
|
||||
const openStep = ctx.sessions.create(SessionId('retry-invariant-open-step'))
|
||||
openStep.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
openStep.append('step/start', { turn: 1, step: 1 })
|
||||
const closedStep = openStep(ctx, 'retry-invariant-closed-step')
|
||||
closedStep.append('step/end', { turn: 1, step: 1 })
|
||||
expect(() => {
|
||||
openStep.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
}).toThrow(/step 1 is still open/)
|
||||
closedStep.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
}).toThrow(/inside an open step/)
|
||||
|
||||
const noStep = ctx.sessions.create(SessionId('retry-invariant-no-step'))
|
||||
noStep.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
noStep.append('turn/start', { turn: 1 })
|
||||
expect(() => {
|
||||
noStep.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
}).toThrow(/latest closed step is undefined/)
|
||||
}).toThrow(/inside an open step/)
|
||||
|
||||
const wrongStep = closeStep(ctx, 'retry-invariant-wrong-step')
|
||||
const wrongStep = openStep(ctx, 'retry-invariant-wrong-step')
|
||||
expect(() => {
|
||||
wrongStep.append('llm/retry', { turn: 1, step: 2, ...normal })
|
||||
}).toThrow(/latest closed step is 1/)
|
||||
}).toThrow(/open step is 1\/1/)
|
||||
|
||||
const closedTurn = closeStep(ctx, 'retry-invariant-closed-turn')
|
||||
closedTurn.append('turn/end', { turn: 1, reason: { kind: 'aborted' } })
|
||||
const closedTurn = openStep(ctx, 'retry-invariant-closed-turn')
|
||||
closedTurn.append('step/end', { turn: 1, step: 1 })
|
||||
closedTurn.append('turn/end', { turn: 1, reason: { kind: 'aborted', reason: { kind: 'user' } },
|
||||
})
|
||||
expect(() => {
|
||||
closedTurn.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
}).toThrow(/inside an open turn/)
|
||||
})
|
||||
|
||||
it('rejects a second retry record for the same step', async () => {
|
||||
it('accepts successive retries in one step and rejects skipped numbering', async () => {
|
||||
const ctx = await setup()
|
||||
const session = closeStep(ctx, 'retry-invariant-duplicate')
|
||||
const session = openStep(ctx, 'retry-invariant-number-sequence')
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...normal, retry: 2 })
|
||||
|
||||
expect(() => {
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...normal, retry: 2 })
|
||||
}).toThrow(/duplicates the retry record/)
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...always, retry: 2 })
|
||||
}).toThrow(/must equal provider policy retry 1/)
|
||||
})
|
||||
|
||||
it('binds retry numbering to the provider policy and resets it after success', async () => {
|
||||
it('binds retry numbering to the provider policy and resets it for a new step', async () => {
|
||||
const ctx = await setup()
|
||||
const mismatch = closeStep(ctx, 'retry-invariant-numbering')
|
||||
const mismatch = openStep(ctx, 'retry-invariant-numbering')
|
||||
mismatch.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
mismatch.append('turn/end', { turn: 1, reason: { kind: 'error', step: 1, failure } })
|
||||
mismatch.append('turn/start', { turn: 2, trigger: { kind: 'retry' } })
|
||||
mismatch.append('step/start', { turn: 2, step: 1 })
|
||||
mismatch.append('step/end', { turn: 2, step: 1 })
|
||||
expect(() => {
|
||||
mismatch.append('llm/retry', { turn: 2, step: 1, ...normal, retry: 1 })
|
||||
mismatch.append('llm/retry', { turn: 1, step: 1, ...normal, retry: 1 })
|
||||
}).toThrow(/must equal provider policy retry 2/)
|
||||
|
||||
const reset = closeStep(ctx, 'retry-invariant-reset')
|
||||
const reset = openStep(ctx, 'retry-invariant-reset')
|
||||
reset.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
reset.append('turn/end', { turn: 1, reason: { kind: 'error', step: 1, failure } })
|
||||
reset.append('turn/start', { turn: 2, trigger: { kind: 'retry' } })
|
||||
reset.append('step/start', { turn: 2, step: 1 })
|
||||
reset.append('assistant/message', {
|
||||
turn: 2,
|
||||
step: 1,
|
||||
message: createMessage({
|
||||
role: 'assistant',
|
||||
content: [{ type: 'text', text: 'success' }],
|
||||
source: {
|
||||
kind: 'model',
|
||||
...{ provider: 'mock', model: 'mock' },
|
||||
},
|
||||
}),
|
||||
}, { surfaceOp: 'append' })
|
||||
reset.append('step/end', { turn: 2, step: 1 })
|
||||
reset.append('turn/end', { turn: 2, reason: { kind: 'completed' } })
|
||||
reset.append('turn/start', { turn: 3, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
reset.append('step/start', { turn: 3, step: 1 })
|
||||
reset.append('step/end', { turn: 3, step: 1 })
|
||||
reset.append('step/end', { turn: 1, step: 1 })
|
||||
reset.append('step/start', { turn: 1, step: 2 })
|
||||
expect(() => {
|
||||
reset.append('llm/retry', { turn: 3, step: 1, ...normal })
|
||||
reset.append('llm/retry', { turn: 1, step: 2, ...normal })
|
||||
}).not.toThrow()
|
||||
})
|
||||
|
||||
@@ -262,9 +236,7 @@ describe('llm-retry invariants', () => {
|
||||
appendRetryTurn(nonFailureEnd, 2)
|
||||
|
||||
const missingStart = ctx.sessions.create(SessionId('retry-invariant-missing-start'))
|
||||
missingStart.append('turn/end', {
|
||||
turn: 1,
|
||||
reason: { kind: 'error', step: 1, failure },
|
||||
missingStart.append('turn/end', { turn: 1, reason: { kind: 'error', error: failure },
|
||||
})
|
||||
appendRetryTurn(missingStart, 2)
|
||||
|
||||
@@ -274,7 +246,7 @@ describe('llm-retry invariants', () => {
|
||||
|
||||
it('rejects a provider that does not match the failed request route', async () => {
|
||||
const ctx = await setup()
|
||||
const session = closeStep(ctx, 'retry-invariant-provider')
|
||||
const session = openStep(ctx, 'retry-invariant-provider')
|
||||
expect(() => {
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...always, provider: 'other' })
|
||||
}).toThrow(/does not match the failed request provider mock/)
|
||||
@@ -284,7 +256,7 @@ describe('llm-retry invariants', () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SessionStore)
|
||||
const session = ctx.sessions.create(SessionId('retry-invariant-late'))
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
session.append('llm/retry', { turn: 1, step: 1, ...normal })
|
||||
await ctx.plugin(InvariantService)
|
||||
await expect(ctx.plugin(RetryInvariant)).rejects.toThrow(/inside an open turn/)
|
||||
|
||||
@@ -32,13 +32,12 @@ describe.each(['jsonl', 'sqlite'] as const)('%s retry-event persistence', (kind)
|
||||
const ctx = await backend(kind)
|
||||
try {
|
||||
const session = ctx.sessions.create(SessionId(`retry-${kind}`))
|
||||
session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } })
|
||||
session.append('turn/start', { turn: 1 })
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
session.append('request/header', {
|
||||
header: { config: { provider: 'mock', model: 'mock' } },
|
||||
reason: 'initial',
|
||||
})
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
const event = session.append('llm/retry', {
|
||||
turn: 1,
|
||||
step: 1,
|
||||
@@ -49,13 +48,9 @@ describe.each(['jsonl', 'sqlite'] as const)('%s retry-event persistence', (kind)
|
||||
delayMs: 750,
|
||||
failure: { message: 'provider busy', code: 'RATE_LIMIT', status: 429 },
|
||||
})
|
||||
session.append('turn/end', {
|
||||
turn: 1,
|
||||
reason: {
|
||||
kind: 'error',
|
||||
step: 1,
|
||||
failure: { message: 'provider busy', code: 'RATE_LIMIT', status: 429 },
|
||||
},
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
session.append('turn/end', { turn: 1, reason: { kind: 'error', error: { message: 'provider busy', code: 'RATE_LIMIT', status: 429 },
|
||||
},
|
||||
})
|
||||
|
||||
expect(session.deriveMessages()).toEqual([])
|
||||
|
||||
@@ -149,15 +149,8 @@ function alwaysConfig(backoff: BackoffConfig = {}): AlwaysRetryPolicyConfig {
|
||||
}
|
||||
}
|
||||
|
||||
function waitForIdle(ctx: Context, agent: Agent): Promise<void> {
|
||||
return new Promise((resolve) => {
|
||||
const dispose = ctx.on('agent/status', (subject, status) => {
|
||||
if (subject === agent && status === 'idle') {
|
||||
dispose()
|
||||
resolve()
|
||||
}
|
||||
})
|
||||
})
|
||||
function waitForIdle(_ctx: Context, agent: Agent): Promise<void> {
|
||||
return agent.whenIdle()
|
||||
}
|
||||
|
||||
function waitForRetry(ctx: Context, agent: Agent, retryNumber: number): Promise<Extract<SessionEvent, { type: 'llm/retry' }>> {
|
||||
@@ -180,7 +173,7 @@ afterEach(async () => {
|
||||
})
|
||||
|
||||
describe('provider-routed retry policy', () => {
|
||||
it('records the scheduled delay before opening a fresh request attempt', async () => {
|
||||
it('records the scheduled delay before retrying the request', async () => {
|
||||
vi.useFakeTimers()
|
||||
const adapter = new ScriptedAdapter([
|
||||
new LlmError('busy', 'RATE_LIMIT', { status: 429 }),
|
||||
@@ -219,7 +212,7 @@ describe('provider-routed retry policy', () => {
|
||||
|
||||
expect(adapter.requests).toHaveLength(2)
|
||||
expect(agent.session.events.filter(item => item.type === 'step/start').map(item => item.data))
|
||||
.toEqual([{ turn: 1, step: 1 }, { turn: 2, step: 1 }])
|
||||
.toEqual([{ turn: 1, step: 1 }])
|
||||
expect(agent.session.deriveMessages().at(-1)).toEqual({
|
||||
id: expect.any(String) as unknown,
|
||||
role: 'assistant',
|
||||
@@ -256,7 +249,7 @@ describe('provider-routed retry policy', () => {
|
||||
expect(agent.session.events.filter(event => event.type === 'assistant/message').map(event => ({
|
||||
turn: event.data.turn,
|
||||
step: event.data.step,
|
||||
}))).toEqual([{ turn: 2, step: 1 }])
|
||||
}))).toEqual([{ turn: 1, step: 1 }])
|
||||
expect(agent.session.deriveMessages().at(-1)).toMatchObject({
|
||||
role: 'assistant',
|
||||
content: [{ type: 'text', text: 'recovered' }],
|
||||
@@ -289,14 +282,21 @@ describe('provider-routed retry policy', () => {
|
||||
await vi.advanceTimersByTimeAsync(500)
|
||||
await idle
|
||||
|
||||
const retryEvent = agent.session.events.find(event => event.type === 'llm/retry')
|
||||
const failedChunks = agent.session.events.filter(event =>
|
||||
event.type === 'assistant/chunk' && event.data.turn === 1 && event.data.step === 1,
|
||||
event.type === 'assistant/chunk'
|
||||
&& retryEvent !== undefined
|
||||
&& event.seq < retryEvent.seq,
|
||||
)
|
||||
expect(failedChunks).toHaveLength(6)
|
||||
expect(agent.session.events.filter(event => event.type === 'assistant/message').map(event => ({
|
||||
expect(failedChunks).toHaveLength(7)
|
||||
const assistantMessages = agent.session.events.filter(event => event.type === 'assistant/message')
|
||||
expect(assistantMessages.map(event => ({
|
||||
turn: event.data.turn,
|
||||
step: event.data.step,
|
||||
}))).toEqual([{ turn: 2, step: 1 }])
|
||||
}))).toEqual([{ turn: 1, step: 1 }])
|
||||
expect(failedChunks.every(event =>
|
||||
!assistantMessages[0]?.sourceEventSeqs?.includes(event.seq),
|
||||
)).toBe(true)
|
||||
expect(agent.session.events.some(event => event.type === 'tool/call')).toBe(false)
|
||||
expect(toolExecutions).toBe(0)
|
||||
expect(agent.session.deriveMessages().at(-1)).toMatchObject({
|
||||
@@ -337,7 +337,7 @@ describe('provider-routed retry policy', () => {
|
||||
expect(agent.session.events.filter(event => event.type === 'llm/retry')).toHaveLength(2)
|
||||
expect(agent.session.events.at(-1)).toMatchObject({
|
||||
type: 'turn/end',
|
||||
data: { reason: { kind: 'error', failure: { message: 'busy three', code: 'SERVER' } } },
|
||||
data: { reason: { kind: 'error', error: { message: 'busy three', code: 'SERVER' } } },
|
||||
})
|
||||
})
|
||||
|
||||
@@ -448,10 +448,14 @@ describe('provider-routed retry policy', () => {
|
||||
|
||||
expect(adapter.requests).toHaveLength(0)
|
||||
expect(agent.session.events.some(event => event.type === 'llm/retry')).toBe(false)
|
||||
expect(agent.session.events.at(-1)).toMatchObject({
|
||||
const end = agent.session.events.at(-1)
|
||||
expect(end).toMatchObject({
|
||||
type: 'turn/end',
|
||||
data: { reason: { kind: 'error', failure: { code: 'NO_ADAPTER' } } },
|
||||
data: { reason: { kind: 'error', error: { code: 'NO_ADAPTER' } } },
|
||||
})
|
||||
if (end?.type === 'turn/end' && end.data.reason.kind === 'error') {
|
||||
expect(end.data.reason.error.message).toContain('no adapter registered for provider')
|
||||
}
|
||||
})
|
||||
|
||||
it('selects policy by the failed request provider', async () => {
|
||||
@@ -502,7 +506,7 @@ describe('provider-routed retry policy', () => {
|
||||
;({ ctx: context } = await harness(adapter, {
|
||||
other: alwaysConfig({ initialDelayMs: 1, maxDelayMs: 1, jitterRatio: 0 }),
|
||||
}, (ctx) => {
|
||||
ctx.on('agent/request', async (_agent, _turn, _step, _signal, next) => ({
|
||||
ctx.on('agent/request', async (_payload, next) => ({
|
||||
...await next(),
|
||||
provider: 'other',
|
||||
}))
|
||||
@@ -539,9 +543,9 @@ describe('provider-routed retry policy', () => {
|
||||
backoff: { initialDelayMs: 1, maxDelayMs: 1 },
|
||||
}),
|
||||
}, (ctx) => {
|
||||
ctx.on('agent/request', async (_agent, turn, _step, _signal, next) => ({
|
||||
ctx.on('agent/request', async (_payload, next) => ({
|
||||
...await next(),
|
||||
provider: turn === 1 ? 'mock' : 'other',
|
||||
provider: adapter.requests.length === 0 ? 'mock' : 'other',
|
||||
}))
|
||||
}))
|
||||
const agent = context.agentLoop.create(SessionId('retry-provider-budgets'), {
|
||||
@@ -877,7 +881,7 @@ describe('provider-routed retry policy', () => {
|
||||
context = mounted.ctx
|
||||
const downstream = Promise.withResolvers<RequestErrorAction>()
|
||||
const entered = Promise.withResolvers<undefined>()
|
||||
context.on('agent/request-error', (agent) => {
|
||||
context.on('agent/request-error', ({ agent }) => {
|
||||
agent.cancel({ kind: 'user' })
|
||||
entered.resolve(undefined)
|
||||
return downstream.promise
|
||||
@@ -913,9 +917,7 @@ describe('provider-routed retry policy', () => {
|
||||
const captured = Promise.withResolvers<undefined>()
|
||||
let invokeCaptured: (() => Promise<void>) | undefined
|
||||
const mounted = await harness(adapter, {}, (ctx) => {
|
||||
ctx.on('agent/request-error', (
|
||||
_agent, _turn, _step, _error, _failure, _history, _retryPolicy, _signal, next,
|
||||
) => {
|
||||
ctx.on('agent/request-error', (_payload, next) => {
|
||||
return new Promise<RequestErrorAction>((resolve) => {
|
||||
invokeCaptured = async () => { resolve(await next()) }
|
||||
captured.resolve(undefined)
|
||||
@@ -924,9 +926,7 @@ describe('provider-routed retry policy', () => {
|
||||
})
|
||||
context = mounted.ctx
|
||||
let downstreamCalls = 0
|
||||
context.on('agent/request-error', async (
|
||||
_agent, _turn, _step, _error, _failure, _history, _retryPolicy, _signal, next,
|
||||
) => {
|
||||
context.on('agent/request-error', async (_payload, next) => {
|
||||
downstreamCalls += 1
|
||||
return next()
|
||||
})
|
||||
@@ -980,9 +980,7 @@ describe('provider-routed retry policy', () => {
|
||||
textResponse('must not run'),
|
||||
])
|
||||
;({ ctx: context } = await harness(adapter, { mock: policy }, (ctx) => {
|
||||
ctx.on('agent/request-error', async (
|
||||
agent, _turn, _step, _error, _failure, _history, _retryPolicy, _signal, next,
|
||||
) => {
|
||||
ctx.on('agent/request-error', async ({ agent }, next) => {
|
||||
agent.cancel({ kind: 'user' })
|
||||
return next()
|
||||
})
|
||||
|
||||
@@ -55,14 +55,8 @@ async function harness(
|
||||
return ctx
|
||||
}
|
||||
|
||||
function waitForIdle(ctx: Context, agent: Agent): Promise<void> {
|
||||
return new Promise((resolve) => {
|
||||
const dispose = ctx.on('agent/status', (subject, status) => {
|
||||
if (subject !== agent || status !== 'idle') return
|
||||
dispose()
|
||||
resolve()
|
||||
})
|
||||
})
|
||||
function waitForIdle(_ctx: Context, agent: Agent): Promise<void> {
|
||||
return agent.whenIdle()
|
||||
}
|
||||
|
||||
function sendAndWait(ctx: Context, agent: Agent): Promise<void> {
|
||||
@@ -109,15 +103,15 @@ describe('bounded retry through the real DeepSeek HTTP/SSE adapter', () => {
|
||||
expect(server?.requests).toHaveLength(1)
|
||||
expect(agent.session.events.filter(event => event.type === 'step/start')
|
||||
.map(event => [event.data.turn, event.data.step]))
|
||||
.toEqual([[1, 1], [2, 1]])
|
||||
.toEqual([[1, 1]])
|
||||
expect(agent.session.events.filter(event => event.type === 'llm/retry').map(event => event.data.failure.code))
|
||||
.toEqual(['TRANSPORT'])
|
||||
expect(finalAssistantText(agent)).toBe('connected after retry')
|
||||
})
|
||||
|
||||
it.each([
|
||||
['stream_disconnect', 0] as const,
|
||||
['partial_disconnect', 2] as const,
|
||||
['stream_disconnect', 1] as const,
|
||||
['partial_disconnect', 3] as const,
|
||||
])('retries %s without committing failed chunks', async (behavior, failedChunkCount) => {
|
||||
const server = await start([behavior, 'success'], {
|
||||
apiKey: 'mock-key',
|
||||
@@ -136,12 +130,15 @@ describe('bounded retry through the real DeepSeek HTTP/SSE adapter', () => {
|
||||
|
||||
expect(server.requests).toHaveLength(2)
|
||||
expect(server.requests[0]?.body).toEqual(server.requests[1]?.body)
|
||||
const retryEvent = agent.session.events.find(event => event.type === 'llm/retry')
|
||||
expect(agent.session.events.filter(event =>
|
||||
event.type === 'assistant/chunk' && event.data.turn === 1,
|
||||
event.type === 'assistant/chunk'
|
||||
&& retryEvent !== undefined
|
||||
&& event.seq < retryEvent.seq,
|
||||
)).toHaveLength(failedChunkCount)
|
||||
expect(agent.session.events.filter(event => event.type === 'assistant/message')
|
||||
.map(event => [event.data.turn, event.data.step]))
|
||||
.toEqual([[2, 1]])
|
||||
.toEqual([[1, 1]])
|
||||
expect(agent.session.events.filter(event => event.type === 'llm/retry').map(event => event.data.failure.code))
|
||||
.toEqual(['TRANSPORT'])
|
||||
expect(finalAssistantText(agent)).toBe('recovered response')
|
||||
@@ -166,7 +163,7 @@ describe('bounded retry through the real DeepSeek HTTP/SSE adapter', () => {
|
||||
.toEqual(['EMPTY_RESPONSE'])
|
||||
expect(agent.session.events.filter(event => event.type === 'assistant/message')
|
||||
.map(event => [event.data.turn, event.data.step]))
|
||||
.toEqual([[2, 1]])
|
||||
.toEqual([[1, 1]])
|
||||
expect(agent.session.events.at(-1)).toMatchObject({
|
||||
type: 'turn/end',
|
||||
data: { reason: { kind: 'completed' } },
|
||||
@@ -191,12 +188,12 @@ describe('bounded retry through the real DeepSeek HTTP/SSE adapter', () => {
|
||||
expect(server.requests).toHaveLength(1)
|
||||
expect(agent.session.events.filter(event =>
|
||||
event.type === 'assistant/chunk' && event.data.turn === 1,
|
||||
)).toHaveLength(2)
|
||||
)).toHaveLength(3)
|
||||
expect(agent.session.events.some(event => event.type === 'assistant/message')).toBe(false)
|
||||
expect(agent.session.events.some(event => event.type === 'llm/retry')).toBe(false)
|
||||
expect(agent.session.events.at(-1)).toMatchObject({
|
||||
type: 'turn/end',
|
||||
data: { reason: { kind: 'error', failure: { code: 'STREAM_CLOSED' } } },
|
||||
data: { reason: { kind: 'error', error: { message: 'SSE stream ended without [DONE]', code: 'STREAM_CLOSED' } } },
|
||||
})
|
||||
})
|
||||
|
||||
@@ -234,11 +231,15 @@ describe('bounded retry through the real DeepSeek HTTP/SSE adapter', () => {
|
||||
await sendAndWait(context, agent)
|
||||
|
||||
expect(server.requests).toHaveLength(3)
|
||||
expect(agent.session.events.filter(event => event.type === 'step/start')).toHaveLength(3)
|
||||
expect(agent.session.events.filter(event => event.type === 'step/start')).toHaveLength(1)
|
||||
expect(agent.session.events.filter(event => event.type === 'llm/retry')).toHaveLength(2)
|
||||
expect(agent.session.events.at(-1)).toMatchObject({
|
||||
const end = agent.session.events.at(-1)
|
||||
expect(end).toMatchObject({
|
||||
type: 'turn/end',
|
||||
data: { reason: { kind: 'error', failure: { code: 'TRANSPORT' } } },
|
||||
data: { reason: { kind: 'error', error: { code: 'TRANSPORT' } } },
|
||||
})
|
||||
if (end?.type === 'turn/end' && end.data.reason.kind === 'error') {
|
||||
expect(end.data.reason.error.message).toContain('DeepSeek API request to')
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/llm/README.md
|
||||
README.md: 901378a17c1a946715cd7181527fed07cf46e447
|
||||
README.zh.md: 3aed67e518a2e9575954595a8a844bfdf8d6a152
|
||||
README.md: d15b2c996d6d47371a3d6c5542eb5c253029dbae
|
||||
README.zh.md: d965f15298f09ff9c2a953a69346c4e83136934b
|
||||
|
||||
@@ -12,16 +12,21 @@ An adapter registry plus a single streaming call surface, interceptable via a wa
|
||||
|
||||
- `ctx.llm.registerAdapter(providers: string[], adapter: LlmAdapter): AdapterRegistrationHandle` Register one adapter instance for the given provider routes. Registration is all-or-nothing, and is disposed with the calling fiber. The returned disposer also carries `replace(providers)`: the candidate route set is validated in full before anything moves, so a conflict with another adapter leaves the current routes registered and serving, and the swap itself is one synchronous section with no observable gap. `replace([])` is legal — a registration holding zero routes — unlike an empty initial registration.
|
||||
- `ctx.llm.listProviders(): LlmProviderInfo[]` Describe registered provider routes in registration order.
|
||||
- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void` Declare provider routes an adapter plugin can activate through configuration — registered or dormant — each naming its owning settings namespace and the path to its profile inside that section. All-or-nothing (`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`), disposed with the calling fiber.
|
||||
- `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` List the declared directory in declaration order; configuration surfaces merge it with `listProviders()` to mark each entry live or dormant.
|
||||
- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle` Declare provider routes an adapter plugin can activate through configuration — registered or dormant — each naming its owning settings namespace and the path to its profile inside that section. All-or-nothing (`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`), disposed with the calling fiber. The handle also carries `replace(entries)`: the candidate set is validated in full before anything moves, so an entry another registration already declares leaves the current set intact, and an empty array is legal there. A plugin whose declared set follows its configuration must use `replace` rather than disposing and re-registering — the latter strands the directory empty whenever the new set is refused.
|
||||
- `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` List the declared directory in declaration order; configuration surfaces merge it with `listProviders()` to mark each entry live or dormant. An entry may carry `declared` — whether the owning adapter knows that route only because configuration named it. Only the adapter can answer, so absence means "this adapter draws no such distinction", never "shipped".
|
||||
- `ctx.llm.registerModelDiscovery(settingsNs: string, discover): () => void` Offer to interrogate provider endpoints for the settings namespace this plugin owns. One offer per namespace (`INVALID_DISCOVERY`/`DUPLICATE_DISCOVERY`), disposed with the calling fiber.
|
||||
- `ctx.llm.listModelDiscoveryNamespaces(): string[]` List the namespaces that can interrogate an endpoint, so a surface offers the action only where it works.
|
||||
- `ctx.llm.discoverModels(settingsNs: string, request: LlmModelDiscoveryRequest): Promise<LlmDiscoveredModel[]>` Ask one endpoint which models it advertises.
|
||||
- `ctx.llm.providerRetryPolicy(provider: string): ResolvedRetryPolicy` Return the provider-owned retry policy captured during registration, with normal defaults resolved.
|
||||
- `ctx.llm.listModels(provider: string): Promise<LlmModelInfo[]>` Discover the models one registered provider currently advertises.
|
||||
- `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<LlmResolvedModelInfo>` Resolve validated exact-model identity plus available context, output-default, and reasoning metadata from the owning adapter, with optional cancellation for asynchronous adapters.
|
||||
- `ctx.llm.resolveCallConfig(config: LlmCallConfig, signal?: AbortSignal): Promise<LlmCallConfig>` Validate an explicit effort and materialize adapter-configured call defaults without clamping.
|
||||
- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise<PreparedLlmCall>` Resolve a config plus detached context metadata and adapter-default provenance in one exact-model lookup, then capture its current adapter registration as one cancellable, one-shot call.
|
||||
- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise<PreparedLlmCall>` Resolve a config plus detached context metadata and adapter-default provenance in one exact-model lookup, then capture its current adapter registration and immutable retry policy as one cancellable, one-shot call.
|
||||
- `ctx.llm.stream(options: GenerateOptions): AsyncIterable<StreamChunk>` Stream one model call as raw chunks (token-level deltas). Consumers assemble the chunks into blocks/messages with `BlockAssembler`.
|
||||
|
||||
`LlmService` preserves errors from final adapter selection, synchronous dispatch, iterator construction, and iteration, and binds their provenance to the exact stream handle returned for that model call. `isLlmAdapterFailure(stream, value)` reports only errors from that call's final adapter boundary; `llmFailureOf(stream, value)` returns the adjacent immutable `LlmFailure`; `llmRetryPolicyOf(stream)` returns the immutable policy of the exact registration selected at that boundary, even if the route is later disposed or replaced. A call that never reaches a final adapter has no serving policy. Nested model calls, `llm/stream` middleware, and downstream consumer failures remain unclassified for the outer call. Classification never replaces or mutates the adapter's original coded `Error`.
|
||||
`LlmService` normalizes failures from final adapter selection, synchronous dispatch, iterator construction, and iteration into the stream protocol's single terminal form: `finish { kind: 'error' | 'aborted', failure }`. A failure after partial deltas may leave content blocks open; consumers discard that incomplete output. Errors from `llm/stream` middleware, nested calls, adapter cleanup, and downstream consumers remain thrown because they are plugin or consumer failures rather than model-request outcomes. A prepared call exposes the immutable retry policy captured with its exact adapter registration; a route handled entirely by middleware has no serving policy.
|
||||
|
||||
Interrogating an endpoint is configuration-time work over a *draft*, which is why it is keyed by settings namespace rather than by provider route: the provider a surface is adding does not exist yet, so there is no route to name. The request may still *name* a route it is editing, and an adapter that already describes that route should answer from its own knowledge — better metadata, no network call — which is why `baseURL` is optional and one of the two is required. The request otherwise carries the endpoint, the protocol, and a credential the harness uses for that one interrogation and never stores — nothing here reads or writes settings or credentials, and the reply is candidate metadata a surface may offer for adoption, never a registered catalog. `LlmDiscoveredModel` makes every field but `id` optional because most provider listings disclose an id and nothing else; a surface adopting one still owes the capacities its adapter requires. Duplicate and unusable ids are dropped, an unserved namespace fails with `NO_DISCOVERY`, and a request naming neither a route nor an endpoint fails with `INVALID_DISCOVERY`.
|
||||
|
||||
Provider and model metadata is a discovery surface, not a routing whitelist. `registerAdapter()` still owns provider exclusivity and captures the adapter's retry policy for each route, while an adapter may accept model ids absent from `listModels()`; consumers must not reject a request because its model is unlisted. Returned selector metadata is detached and invalid or duplicate adapter entries fail with `INVALID_ADAPTER` or `INVALID_CATALOG`.
|
||||
|
||||
@@ -35,7 +40,6 @@ Exact-model metadata is a separate correctness query, not a catalog decoration o
|
||||
|
||||
| Event | Mode | Purpose |
|
||||
|---|---|---|
|
||||
| `llm/adapters-updated` | emit | Notify consumers to re-read provider and model topology after a registry commit |
|
||||
| `llm/stream` | waterfall | Intercept/wrap every streaming model call for caching, logging, or routing |
|
||||
|
||||
### Extension points
|
||||
@@ -49,7 +53,7 @@ Exact-model metadata is a separate correctness query, not a catalog decoration o
|
||||
|
||||
Message content is an array of typed blocks: `text`, `reasoning`, `tool-call`, `tool-result`. The union is derived from the merge-extensible `ContentBlockMap`, so plugins can add block types via declaration merging. Assistant messages use a model source carrying provider/model provenance and optional adapter-private replay state. Before dispatch, `LlmService` retains that state only when the historical provider route and target provider route are currently owned by the exact same adapter instance; the adapter then decides whether it can restore or convert the state across models/providers. The core block set is limited to blocks every shipping path honors — multimodal content (images, audio, …) has no core block type; a feature that needs one adds it via the map together with the adapter/UI/compaction support that honors it.
|
||||
|
||||
Streaming is a raw chunk protocol (`block-start`, `text-delta`, `reasoning-delta`, `tool-call-delta`, `block-end`, `usage`, `finish`). `BlockAssembler` is the single shared implementation that assembles chunks into blocks/messages.
|
||||
Streaming is a raw chunk protocol (`block-start`, `text-delta`, `reasoning-delta`, `tool-call-delta`, `block-end`, `usage`, `finish`). Every adapter outcome reaches consumers as one terminal `finish`; operational failure uses its `error` or `aborted` reason rather than throwing across the stream API. `BlockAssembler` is the single shared implementation that assembles chunks into blocks/messages.
|
||||
|
||||
### Call configuration (`call-config.ts`)
|
||||
|
||||
@@ -59,6 +63,10 @@ Streaming is a raw chunk protocol (`block-start`, `text-delta`, `reasoning-delta
|
||||
|
||||
Every product adapter sends application identity on provider HTTP requests. `attributionHeaders(identity?)` builds the standard `User-Agent`, defaulting to public `APP_IDENTITY`; white-label deployments may replace but not suppress it. Adapters verify the wire header directly or through their library hook. See [the attribution Agent Note](../../../.agents/notes/implemented/architecture/2026-06-21-mandatory-app-attribution-headers.md).
|
||||
|
||||
### API key validation (`api-key.ts`)
|
||||
|
||||
Every adapter that puts a credential in an HTTP header judges it the same way before use. `normalizeApiKey(raw)` trims surrounding whitespace, then accepts any non-empty printable-ASCII value (`/^[\x21-\x7E]+$/`, space excluded) or reports why not as an `ApiKeyRejection` (`'empty'` | `'illegalCharacters'`), both carried in the `ApiKeyCheck` result. Absence is never judged: a caller decides whether a value was supplied before asking, since a profile naming no credential authenticates through the provider's own ambient discovery or OAuth.
|
||||
|
||||
### Classes
|
||||
|
||||
- `LlmAdapter` — abstract base class for provider adapters. The only required method is `stream()`.
|
||||
@@ -69,10 +77,11 @@ Every product adapter sends application identity on provider HTTP requests. `att
|
||||
- `CONTEXT_WINDOW_EXCEEDED_CODE` — the provider-neutral code both DeepSeek adapters use when a request exceeds the model context window, regardless of thrown-HTTP versus in-band finish delivery. `isContextWindowExceededError(detail)` is their shared conservative classifier for OpenAI-compatible provider detail.
|
||||
- `QUOTA_EXCEEDED_CODE` — the non-transient provider-neutral code for exhausted account quota, balance, credits, budget, or usage limits. `isQuotaExceededError(detail)` keeps those failures distinct from request-rate limits.
|
||||
- `EMPTY_RESPONSE_CODE` — the provider-neutral code both adapters use for a degenerate provider completion: a terminal `stop` that carried no content blocks at all. Classified as an error finish (not a successful empty message) because the attempt produced nothing durable; `dsh-llm-retry` retries it by default.
|
||||
- `INVALID_CREDENTIAL_CODE` — the provider-neutral code for a credential that was supplied but cannot be used: malformed rather than absent, so the fix is to correct the stored value rather than supply one — the distinction from `MISSING_CREDENTIAL`. Deliberately excluded from the default retryable set, since a malformed credential fails identically on every attempt. `assertUsableApiKey(raw, pkg, ref)` throws `LlmError` with this code, the one shared diagnosis every adapter uses for an unusable stored credential.
|
||||
|
||||
### Real adapters
|
||||
|
||||
Two adapters implement `LlmAdapter` on different internals: [`@deepseek-ai/dsh-llm-deepseek`](../llm-deepseek) uses direct fetch with `eventsource-parser` SSE framing for the `deepseek-official` route, while [`@deepseek-ai/dsh-llm-pi-ai`](../llm-pi-ai) dynamically resolves configured provider/model pairs through `@earendil-works/pi-ai`. Both follow the `StreamChunk` conventions in `types.ts`: usage precedes finish, tool arguments remain raw strings, and errors take one of two sanctioned paths. See [the twin LLM adapters](../../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md) for the design rationale.
|
||||
Two adapters implement `LlmAdapter` on different internals: [`@deepseek-ai/dsh-llm-deepseek`](../llm-deepseek) uses direct fetch with `eventsource-parser` SSE framing for the `deepseek-official` route, while [`@deepseek-ai/dsh-llm-pi-ai`](../llm-pi-ai) dynamically resolves configured provider/model pairs through `@earendil-works/pi-ai`. Both follow the `StreamChunk` conventions in `types.ts`: usage precedes finish and tool arguments remain raw strings. Adapter implementations may throw or emit a failure finish internally; `LlmService` exposes both as a terminal failure finish. See [the twin LLM adapters](../../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md) for the adapter rationale and [the terminal-failure decision](../../../.agents/notes/implemented/architecture/2026-07-29-terminal-llm-stream-failures.md) for the service boundary.
|
||||
|
||||
## Model Experience
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
[English](README.md) | 中文
|
||||
|
||||
提供方无关的 LLM(大语言模型)词汇与抽象服务。本包(package)定义 agent loop(智能体循环)、会话日志和每个插件使用的规范语言。
|
||||
提供方无关的 LLM(大语言模型)词汇与抽象服务。本包定义 agent loop(智能体循环)、会话日志和每个插件使用的规范语言。
|
||||
|
||||
## 服务:`LlmService`(ctx key:`llm`)
|
||||
|
||||
@@ -12,16 +12,21 @@
|
||||
|
||||
- `ctx.llm.registerAdapter(providers: string[], adapter: LlmAdapter): AdapterRegistrationHandle` 为给定提供方路由注册一个适配器实例。注册要么全部成功,要么全部不生效,并且会随调用 fiber 一起 dispose(资源释放)。返回的释放器还携带 `replace(providers)`:候选路由集合会在任何东西变动之前完整校验,因此与另一适配器冲突时,当前路由保持注册且继续服务,而替换本身是一个同步区段,不存在可观察的空档。`replace([])` 合法——一个持有零条路由的注册——这与空的初始注册不同。
|
||||
- `ctx.llm.listProviders(): LlmProviderInfo[]` 按注册顺序描述已注册提供方路由。
|
||||
- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void` 声明适配器插件可通过配置激活的提供方路由——无论已注册还是休眠——每个条目指明其所属 settings namespace,以及 profile 在该分节内的路径。要么全部成功,要么全部不生效(`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`),并随调用 fiber dispose。
|
||||
- `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` 按声明顺序列出已声明的目录;配置界面将其与 `listProviders()` 合并,为每个条目标注存活或休眠。
|
||||
- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle` 声明适配器插件可通过配置激活的提供方路由——无论已注册还是休眠——每个条目指明其所属 settings namespace,以及 profile 在该分节内的路径。要么全部成功,要么全部不生效(`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`),并随调用 fiber dispose。该句柄还带 `replace(entries)`:候选集合会先被整体校验,因此其中若有条目已被另一个注册声明,当前集合原封不动;此处允许传空数组。声明集合随配置变化的插件必须使用 `replace`,而不是先 dispose 再重新注册——后者会在新集合被拒时让目录整个落空。
|
||||
- `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` 按声明顺序列出已声明的目录;配置界面将其与 `listProviders()` 合并,为每个条目标注存活或休眠。条目可携带 `declared`——拥有该路由的适配器是否只因配置点名才知道它。只有适配器能回答,因此缺席意味着「这个适配器不作此区分」,绝不等于「内置」。
|
||||
- `ctx.llm.registerModelDiscovery(settingsNs: string, discover): () => void` 为本插件拥有的 settings namespace 提供「询问提供方端点」的能力。每个 namespace 只能有一个(`INVALID_DISCOVERY`/`DUPLICATE_DISCOVERY`),并随调用 fiber dispose。
|
||||
- `ctx.llm.listModelDiscoveryNamespaces(): string[]` 列出可以询问端点的 namespace,让界面只在可用之处提供该动作。
|
||||
- `ctx.llm.discoverModels(settingsNs: string, request: LlmModelDiscoveryRequest): Promise<LlmDiscoveredModel[]>` 询问某个端点它公布了哪些模型。
|
||||
- `ctx.llm.providerRetryPolicy(provider: string): ResolvedRetryPolicy` 返回注册时捕获的提供方重试策略,并解析 normal 默认值。
|
||||
- `ctx.llm.listModels(provider: string): Promise<LlmModelInfo[]>` 发现某个已注册提供方当前公布的模型。
|
||||
- `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<LlmResolvedModelInfo>` 从拥有精确路由的适配器解析经校验的确切模型身份,以及可用上下文、输出默认值和推理(reasoning)元数据;异步适配器可选地支持取消。
|
||||
- `ctx.llm.resolveCallConfig(config: LlmCallConfig, signal?: AbortSignal): Promise<LlmCallConfig>` 校验显式推理强度,并填入适配器配置的调用默认值,但不自动调整。
|
||||
- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise<PreparedLlmCall>` 在一次精确模型查询中解析配置、脱耦的上下文元数据与适配器默认值溯源,再将其当前适配器注册捕获为一次可取消、一次性调用。
|
||||
- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise<PreparedLlmCall>` 在一次精确模型查询中解析配置、脱耦的上下文元数据与适配器默认值溯源,再将当前适配器注册和不可变重试策略捕获为一次可取消、一次性调用。
|
||||
- `ctx.llm.stream(options: GenerateOptions): AsyncIterable<StreamChunk>` 将一次模型调用流式输出为原始分片(token 级增量)。消费方使用 `BlockAssembler` 将分片组装为块/消息。
|
||||
|
||||
`LlmService` 保留来自最终适配器选择、同步 dispatch、iterator 构造与迭代的错误,并将其溯源绑定到该次模型调用返回的精确流句柄。`isLlmAdapterFailure(stream, value)` 只报告该调用最终适配器边界的错误;`llmFailureOf(stream, value)` 返回关联的不可变 `LlmFailure`;`llmRetryPolicyOf(stream)` 返回在该边界选中的确切注册所对应的不可变策略,即使之后释放或替换路由也不变。未到达最终适配器的调用没有服务策略。嵌套模型调用、`llm/stream` middleware 和下游消费方失败对外层调用仍未分类。分类绝不替换或更改适配器原有的带代码 `Error`。
|
||||
`LlmService` 将最终适配器选择、同步 dispatch、iterator 构造与迭代中的失败规范化为流协议唯一的终止形式:`finish { kind: 'error' | 'aborted', failure }`。部分增量输出后发生失败时,内容块可能仍未闭合;消费方会丢弃这些不完整输出。`llm/stream` middleware、嵌套调用、适配器清理和下游消费方的错误仍会抛出,因为它们属于插件或消费方失败,而非模型请求结果。已准备调用会暴露随其确切适配器注册一同捕获的不可变重试策略;完全由 middleware 处理的路由没有服务策略。
|
||||
|
||||
询问端点属于配置期针对**草稿**的操作,因此以 settings namespace 而非提供方路由为键:界面正在新增的提供方还不存在,也就没有路由可点名。但请求仍可**点名**它正在编辑的路由,而已经描述该路由的适配器应当用自己的知识作答——元数据更好,且无需联网——这正是 `baseURL` 可选、两者必居其一的原因。除此之外,请求携带端点、协议,以及一条 harness 只用于这一次询问、绝不存储的凭据——这里既不读也不写 settings 与 credentials,回复是界面可供用户采纳的候选元数据,而不是已注册的 catalog。`LlmDiscoveredModel` 除 `id` 外每个字段都是可选的,因为大多数提供方列表只公布 id;采纳其中一条的界面仍要补上其适配器所需的容量。重复与不可用的 id 会被丢弃,无人服务的 namespace 以 `NO_DISCOVERY` 失败,既不点名路由也不给端点的请求以 `INVALID_DISCOVERY` 失败。
|
||||
|
||||
提供方与模型元数据是发现接口,不是路由白名单。`registerAdapter()` 仍拥有提供方排他性,并为每条路由捕获适配器的重试策略;适配器则可以接受 `listModels()` 中不存在的模型 id,消费方禁止因模型未列出而拒绝请求。返回的 selector 元数据与输入脱离,无效或重复适配器配置项会以 `INVALID_ADAPTER` 或 `INVALID_CATALOG` 失败。
|
||||
|
||||
@@ -35,7 +40,6 @@
|
||||
|
||||
| 事件 | 模式 | 用途 |
|
||||
|---|---|---|
|
||||
| `llm/adapters-updated` | emit | 在注册表提交后通知消费方重新读取提供方和模型拓扑 |
|
||||
| `llm/stream` | waterfall | 拦截/包装每次流式模型调用,用于缓存、日志或路由 |
|
||||
|
||||
### 扩展点
|
||||
@@ -49,7 +53,7 @@
|
||||
|
||||
消息内容是类型化内容块数组:`text`、`reasoning`、`tool-call`、`tool-result`。联合从可合并扩展的 `ContentBlockMap` 派生,因此插件可以通过 declaration merging 添加块类型。assistant 消息使用模型来源,其中携带提供方/模型溯源与可选适配器私有回放状态。dispatch 前,`LlmService` 只在历史提供方路由与目标提供方路由当前由完全相同的适配器实例拥有时才保留该状态;随后由适配器判定能否在模型/提供方间恢复或转换该状态。核心块集只包含每条已发布路径都支持的块。多模态内容(图像、音频等)没有核心块类型;需要它的功能会通过 map 添加,并一并添加相应的适配器/UI/压缩(compaction)支持。
|
||||
|
||||
流式输出是原始分片协议(`block-start`、`text-delta`、`reasoning-delta`、`tool-call-delta`、`block-end`、`usage`、`finish`)。`BlockAssembler` 是将分片组装为块/消息的唯一共享实现。
|
||||
流式输出是原始分片协议(`block-start`、`text-delta`、`reasoning-delta`、`tool-call-delta`、`block-end`、`usage`、`finish`)。每个适配器结果都以一个终止 `finish` 到达消费方;运行故障使用其 `error` 或 `aborted` 原因,而不会跨流 API 抛出。`BlockAssembler` 是将分片组装为块/消息的唯一共享实现。
|
||||
|
||||
### 调用配置(`call-config.ts`)
|
||||
|
||||
@@ -57,7 +61,11 @@
|
||||
|
||||
### 应用归因(`attribution.ts`)
|
||||
|
||||
每个产品适配器都会在提供方 HTTP 请求上发送应用身份。`attributionHeaders(identity?)` 构建标准 `User-Agent`,默认为公开 `APP_IDENTITY`;白标部署可以替换它,但不能抑制它。适配器会直接验证 wire 标头,或通过自身库 hook 验证。详见 [归因 Agent Note(agent 决策记录)](../../../.agents/notes/implemented/architecture/2026-06-21-mandatory-app-attribution-headers.md)。
|
||||
每个产品适配器都会在提供方 HTTP 请求上发送应用身份。`attributionHeaders(identity?)` 构建标准 `User-Agent`,默认为公开 `APP_IDENTITY`;白标部署可以替换它,但不能抑制它。适配器会直接验证 wire 标头,或通过自身库 hook 验证。详见 [归因 Agent Note](../../../.agents/notes/implemented/architecture/2026-06-21-mandatory-app-attribution-headers.md)。
|
||||
|
||||
### API 密钥校验(`api-key.ts`)
|
||||
|
||||
每个要把凭据放进 HTTP 标头的适配器,使用前都以同一套规则校验它。`normalizeApiKey(raw)` 先去除首尾空白,再接受任意非空的可打印 ASCII 值(`/^[\x21-\x7E]+$/`,不含空格),否则以 `ApiKeyRejection`(`'empty'` | `'illegalCharacters'`)说明拒绝原因,二者一并包含在 `ApiKeyCheck` 结果中。缺失从不参与校验:调用方会在询问之前自行判断是否提供了值——未点名凭据的 profile 会转由提供方自身的环境发现或 OAuth 完成认证。
|
||||
|
||||
### 类
|
||||
|
||||
@@ -69,10 +77,11 @@
|
||||
- `CONTEXT_WINDOW_EXCEEDED_CODE`:当请求超过模型上下文窗口时,无论通过 HTTP 异常抛出还是带内 finish 交付,两个 DeepSeek 适配器都使用的提供方无关 code。`isContextWindowExceededError(detail)` 是它们针对 OpenAI 兼容提供方详细信息的共享保守分类器。
|
||||
- `QUOTA_EXCEEDED_CODE`:帐户配额、余额、点数、预算或用量限制耗尽时使用的非短暂提供方无关 code。`isQuotaExceededError(detail)` 使这些失败与请求速率限制保持区分。
|
||||
- `EMPTY_RESPONSE_CODE`:两个适配器都使用的提供方无关 code,用于表示退化的提供方生成结果:一个未携带任何内容块的终止 `stop`。它会被分类为错误 finish(而非成功空消息),因为尝试未产生持久内容;`dsh-llm-retry` 默认重试它。
|
||||
- `INVALID_CREDENTIAL_CODE`:已提供但无法使用的凭据所用的提供方无关 code——格式错误而非缺失,修复方式是改正已存储的值而非补供一个,这正是它与 `MISSING_CREDENTIAL` 的区别。它被刻意排除在默认可重试集合之外:格式错误的凭据每次尝试都会以同样方式失败。`assertUsableApiKey(raw, pkg, ref)` 会以该 code 抛出 `LlmError`,是每个适配器判定已存储凭据不可用时共用的诊断。
|
||||
|
||||
### 真实适配器
|
||||
|
||||
两个适配器使用不同内部机制实现 `LlmAdapter`:[`@deepseek-ai/dsh-llm-deepseek`](../llm-deepseek) 针对 `deepseek-official` 路由使用直接 fetch 加 `eventsource-parser` SSE(Server-Sent Events)分帧,[`@deepseek-ai/dsh-llm-pi-ai`](../llm-pi-ai) 则通过 `@earendil-works/pi-ai` 动态解析已配置提供方/模型对。两者都遵循 `StreamChunk` 约定,定义见 `types.ts`:usage 先于 finish,工具参数保持原始字符串,错误使用两种已批准路径之一。设计理由见 [双 LLM 适配器](../../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md)。
|
||||
两个适配器使用不同内部机制实现 `LlmAdapter`:[`@deepseek-ai/dsh-llm-deepseek`](../llm-deepseek) 针对 `deepseek-official` 路由使用直接 fetch 加 `eventsource-parser` SSE(Server-Sent Events)分帧,[`@deepseek-ai/dsh-llm-pi-ai`](../llm-pi-ai) 则通过 `@earendil-works/pi-ai` 动态解析已配置提供方/模型对。两者都遵循 `types.ts` 中的 `StreamChunk` 约定:usage 先于 finish,工具参数保持原始字符串。适配器实现在内部可以抛出异常或发出失败 finish;`LlmService` 会将两者都暴露为终止失败 finish。适配器理由见[双 LLM 适配器](../../../.agents/notes/implemented/architecture/2026-06-13-twin-llm-adapters.md),服务边界见[终止失败决策](../../../.agents/notes/implemented/architecture/2026-07-29-terminal-llm-stream-failures.md)。
|
||||
|
||||
## 模型体验
|
||||
|
||||
|
||||
@@ -34,9 +34,7 @@
|
||||
"lib/index.js",
|
||||
"lib/invariant.js",
|
||||
"lib/types/**/*.js",
|
||||
"lib/types/**/*.d.ts",
|
||||
"lib/types/**/*.d.ts.map",
|
||||
"src"
|
||||
"lib/types/**/*.d.ts"
|
||||
],
|
||||
"license": "BSD-3-Clause",
|
||||
"peerDependencies": {
|
||||
|
||||
@@ -1,67 +1,40 @@
|
||||
/**
|
||||
* Private provider-failure tagging shared by `LlmService` and its consumers.
|
||||
* Normalization for values thrown by a final LLM adapter boundary.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-llm/adapter-failure
|
||||
*/
|
||||
|
||||
import { HarnessError } from './error.ts'
|
||||
import type { LlmFailure, StreamChunk } from './types.ts'
|
||||
import type { ResolvedRetryPolicy } from './retry-policy.ts'
|
||||
|
||||
/** Call-local facts captured when one model call enters its final adapter boundary. */
|
||||
export interface AdapterFailureScope {
|
||||
/** Errors and normalized facts proven to originate in this call's final adapter boundary. */
|
||||
readonly failures: WeakMap<Error, LlmFailure>
|
||||
/** Immutable policy of the exact adapter registration selected for this call. */
|
||||
retryPolicy?: ResolvedRetryPolicy
|
||||
}
|
||||
|
||||
/** Call-local failure scopes keyed by the exact stream handle returned to a consumer. */
|
||||
const adapterFailureScopes = new WeakMap<AsyncIterable<StreamChunk>, AdapterFailureScope>()
|
||||
import type { LlmFailure } from './types.ts'
|
||||
|
||||
/**
|
||||
* Bind one call's adapter-failure scope to a unique returned stream handle.
|
||||
* @param stream - the waterfall-selected stream for this call.
|
||||
* @param failures - errors tagged by this call's final adapter boundary.
|
||||
* @returns a unique stream handle that delegates iteration to `stream`.
|
||||
* Detach serializable provider facts from a value thrown by an adapter.
|
||||
* @param value - arbitrary value thrown during adapter dispatch or iteration.
|
||||
* @returns immutable provider-neutral facts suitable for a terminal finish chunk.
|
||||
* @internal
|
||||
*/
|
||||
export function bindAdapterFailureScope(
|
||||
stream: AsyncIterable<StreamChunk>,
|
||||
failures: AdapterFailureScope,
|
||||
): AsyncIterable<StreamChunk> {
|
||||
const call = {
|
||||
[Symbol.asyncIterator](): AsyncIterator<StreamChunk> {
|
||||
return stream[Symbol.asyncIterator]()
|
||||
},
|
||||
}
|
||||
adapterFailureScopes.set(call, failures)
|
||||
return call
|
||||
}
|
||||
|
||||
/**
|
||||
* Preserve an adapter's Error identity while tagging its provider origin.
|
||||
* @param failures - the call-local final-adapter failure scope.
|
||||
* @param value - arbitrary value thrown by adapter dispatch or iteration.
|
||||
* @returns the original Error, or a coded Error wrapping a non-Error throw.
|
||||
* @internal
|
||||
*/
|
||||
export function markLlmAdapterFailure(
|
||||
failures: AdapterFailureScope,
|
||||
value: unknown,
|
||||
): Error & { code?: string } {
|
||||
export function normalizeLlmFailure(value: unknown): LlmFailure {
|
||||
const error = value instanceof Error
|
||||
? value as Error & { code?: string }
|
||||
: new HarnessError(String(value), 'UNKNOWN', { cause: value })
|
||||
? value
|
||||
: new HarnessError(thrownMessage(value), 'UNKNOWN', { cause: value })
|
||||
// Cross-package copies preserve own data but not class identity. Trust the
|
||||
// carried facts only when both own properties agree after validation.
|
||||
const carried = ownFailureSnapshot(error)
|
||||
const failure = carried !== undefined && carried.code === ownErrorCode(error) ? carried : Object.freeze({
|
||||
if (carried !== undefined && carried.code === ownErrorCode(error)) return carried
|
||||
return Object.freeze({
|
||||
message: errorMessage(error),
|
||||
code: harnessErrorCode(error),
|
||||
})
|
||||
failures.failures.set(error, failure)
|
||||
return error
|
||||
}
|
||||
|
||||
/** Render a non-Error throw without letting hostile coercion escape normalization. */
|
||||
function thrownMessage(value: unknown): string {
|
||||
try {
|
||||
const message = String(value)
|
||||
return message.length > 0 ? message : 'LLM adapter failed'
|
||||
} catch (_hostileThrownValue) {
|
||||
return 'LLM adapter failed'
|
||||
}
|
||||
}
|
||||
|
||||
/** Read a foreign error's own data-backed `code` without invoking accessors. */
|
||||
@@ -129,46 +102,3 @@ function errorMessage(error: Error): string {
|
||||
function harnessErrorCode(error: Error): string {
|
||||
return error instanceof HarnessError ? error.code : 'UNKNOWN'
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a failure came from final adapter dispatch, iterator construction,
|
||||
* or iteration for the call represented by the exact returned stream handle.
|
||||
* @param stream - the exact stream returned by the model call being classified.
|
||||
* @param value - arbitrary failure caught by a model-call consumer.
|
||||
* @returns true only for errors tagged at that call's final adapter boundary.
|
||||
*/
|
||||
export function isLlmAdapterFailure(
|
||||
stream: AsyncIterable<StreamChunk>,
|
||||
value: unknown,
|
||||
): value is Error & { code?: string } {
|
||||
const failures = adapterFailureScopes.get(stream)
|
||||
return value instanceof Error && failures !== undefined && failures.failures.has(value)
|
||||
}
|
||||
|
||||
/**
|
||||
* Retrieve normalized provider facts only for an Error tagged by this exact
|
||||
* model call's final adapter boundary.
|
||||
* @param stream - the exact stream returned to the consumer.
|
||||
* @param value - the caught failure.
|
||||
* @returns the immutable facts for that call, or `undefined` for middleware, nested, or consumer failures.
|
||||
*/
|
||||
export function llmFailureOf(
|
||||
stream: AsyncIterable<StreamChunk>,
|
||||
value: unknown,
|
||||
): LlmFailure | undefined {
|
||||
const failures = adapterFailureScopes.get(stream)
|
||||
return value instanceof Error ? failures?.failures.get(value) : undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the retry policy of the exact registration selected at this call's
|
||||
* final adapter boundary. The policy remains available after that registration
|
||||
* is disposed or replaced; absence means no final adapter served the call.
|
||||
* @param stream - the exact stream returned by the model call.
|
||||
* @returns the immutable serving-registration policy, or `undefined`.
|
||||
*/
|
||||
export function llmRetryPolicyOf(
|
||||
stream: AsyncIterable<StreamChunk>,
|
||||
): ResolvedRetryPolicy | undefined {
|
||||
return adapterFailureScopes.get(stream)?.retryPolicy
|
||||
}
|
||||
|
||||
41
packages/llm/llm/src/api-key.ts
Normal file
41
packages/llm/llm/src/api-key.ts
Normal file
@@ -0,0 +1,41 @@
|
||||
/**
|
||||
* The one definition of a well-formed provider API key, shared by every
|
||||
* adapter that puts one in an HTTP header.
|
||||
* @module @deepseek-ai/dsh-llm/api-key
|
||||
*/
|
||||
|
||||
/**
|
||||
* Characters an HTTP header value carries verbatim and every known provider
|
||||
* key uses: printable ASCII, space excluded. A key outside this set cannot
|
||||
* reach any provider — `fetch` refuses to build the header — so this is a
|
||||
* transport invariant rather than one provider's policy. Latin-1 is excluded
|
||||
* deliberately: a header could carry it, but no provider issues it, and
|
||||
* admitting it trades a local explained refusal for an opaque 401.
|
||||
*/
|
||||
const LEGAL_API_KEY = /^[\x21-\x7E]+$/
|
||||
|
||||
/** Why a supplied API key cannot be used. */
|
||||
export type ApiKeyRejection = 'empty' | 'illegalCharacters'
|
||||
|
||||
/** The verdict on one supplied API key. */
|
||||
export type ApiKeyCheck =
|
||||
| { readonly ok: true; readonly value: string }
|
||||
| { readonly ok: false; readonly reason: ApiKeyRejection }
|
||||
|
||||
/**
|
||||
* Judge one *supplied* API key, trimming surrounding whitespace first.
|
||||
*
|
||||
* Trimming is silent because a padded key has one unambiguous reading; every
|
||||
* other defect is reported. Absence is a configuration state this function
|
||||
* never sees — a profile naming no credential authenticates through the
|
||||
* provider's own ambient discovery or OAuth — so callers decide whether a
|
||||
* value was supplied before asking.
|
||||
* @param raw - the key exactly as configured, stored, or typed.
|
||||
* @returns the trimmed key, or why it cannot be used.
|
||||
*/
|
||||
export function normalizeApiKey(raw: string): ApiKeyCheck {
|
||||
const value = raw.trim()
|
||||
if (value.length === 0) return { ok: false, reason: 'empty' }
|
||||
if (!LEGAL_API_KEY.test(value)) return { ok: false, reason: 'illegalCharacters' }
|
||||
return { ok: true, value }
|
||||
}
|
||||
@@ -127,11 +127,15 @@ export class BlockAssembler {
|
||||
|
||||
/**
|
||||
* Assemble all blocks seen so far, in stream order.
|
||||
* @returns one block per seen index; an open block assembles from its
|
||||
* accumulated deltas (an unknown block type never closed by `block-end` throws).
|
||||
* @returns one block per seen index, except that max-token truncation drops
|
||||
* tool calls that cannot be executed safely; an open block assembles from
|
||||
* its accumulated deltas (an unknown block type never closed by `block-end` throws).
|
||||
*/
|
||||
blocks(): ContentBlock[] {
|
||||
return this.order.map(index => this.assemble(this.mustGet(index), index))
|
||||
const blocks = this.order.map(index => this.assemble(this.mustGet(index), index))
|
||||
return this.finish.kind === 'max-tokens'
|
||||
? blocks.filter(block => block.type !== 'tool-call')
|
||||
: blocks
|
||||
}
|
||||
|
||||
/** Usage from the `usage` chunk; undefined until one arrives. */
|
||||
|
||||
@@ -38,6 +38,15 @@ export const QUOTA_EXCEEDED_CODE = 'QUOTA'
|
||||
*/
|
||||
export const EMPTY_RESPONSE_CODE = 'EMPTY_RESPONSE'
|
||||
|
||||
/**
|
||||
* Canonical provider-neutral code for a credential that was supplied but
|
||||
* cannot be used — malformed rather than absent. Distinct from
|
||||
* `MISSING_CREDENTIAL` because the fix differs: correct the stored value
|
||||
* rather than supply one. Deliberately outside the default retryable set —
|
||||
* a malformed credential fails identically on every attempt.
|
||||
*/
|
||||
export const INVALID_CREDENTIAL_CODE = 'INVALID_CREDENTIAL'
|
||||
|
||||
/** Structured codes and plain phrases that explicitly name a context bound being exceeded. */
|
||||
const STRUCTURED_CONTEXT_OVERFLOW = new RegExp(
|
||||
String.raw`(?:^|[^a-z0-9])context[\s_-](?:length|window)[\s_-]`
|
||||
@@ -93,7 +102,8 @@ export function isQuotaExceededError(detail: string): boolean {
|
||||
/**
|
||||
* Render a thrown value with its full `cause` chain and AggregateError
|
||||
* members, so transport wrappers like undici's `TypeError: fetch failed`
|
||||
* surface the underlying failure instead of masking it. Diagnostic-surface
|
||||
* surface the underlying failure instead of masking it. Plain structured
|
||||
* failures render their own data-backed `message`. Diagnostic-surface
|
||||
* rendering only (messages, notices, logs) — never parse the result; route on
|
||||
* {@link HarnessError.code}.
|
||||
* @param value - the caught value (`unknown` in catch clauses).
|
||||
@@ -109,7 +119,15 @@ export function errorChain(value: unknown): string {
|
||||
if (path.has(current)) return '<circular cause>'
|
||||
path.add(current)
|
||||
try {
|
||||
if (!(current instanceof Error)) return String(current)
|
||||
if (!(current instanceof Error)) {
|
||||
if (typeof current === 'object' && current !== null) {
|
||||
const descriptor = Object.getOwnPropertyDescriptor(current, 'message')
|
||||
if (descriptor !== undefined && 'value' in descriptor && typeof descriptor.value === 'string') {
|
||||
return descriptor.value
|
||||
}
|
||||
}
|
||||
return String(current)
|
||||
}
|
||||
const message = current.message === '' ? current.name : current.message
|
||||
const members = current instanceof AggregateError && current.errors.length > 0
|
||||
? ` [${current.errors.map(render).join('; ')}]`
|
||||
|
||||
@@ -10,8 +10,10 @@ import { Context, Service } from 'cordis'
|
||||
import type {
|
||||
GenerateOptions,
|
||||
LlmConfigurableProvider,
|
||||
LlmDiscoveredModel,
|
||||
LlmFailure,
|
||||
LlmModelContext,
|
||||
LlmModelDiscoveryRequest,
|
||||
LlmModelInfo,
|
||||
LlmResolvedModelInfo,
|
||||
LlmProviderInfo,
|
||||
@@ -24,14 +26,15 @@ import type { ResolvedRetryPolicy } from './retry-policy.ts'
|
||||
import type { ProviderRequestId } from './brand.ts'
|
||||
import { callConfigEquals, deepFreeze } from './call-config.ts'
|
||||
import type { LlmCallConfig, LlmCallConfigAdapterDefaults } from './call-config.ts'
|
||||
import { HarnessError } from './error.ts'
|
||||
import { bindAdapterFailureScope, markLlmAdapterFailure } from './adapter-failure.ts'
|
||||
import type { AdapterFailureScope } from './adapter-failure.ts'
|
||||
import { HarnessError, INVALID_CREDENTIAL_CODE } from './error.ts'
|
||||
import { normalizeLlmFailure } from './adapter-failure.ts'
|
||||
import { normalizeApiKey } from './api-key.ts'
|
||||
|
||||
export * from './attribution.ts'
|
||||
export * from './brand.ts'
|
||||
export * from './never.ts'
|
||||
export * from './error.ts'
|
||||
export * from './api-key.ts'
|
||||
export * from './types.ts'
|
||||
export * from './content.ts'
|
||||
export * from './message.ts'
|
||||
@@ -39,7 +42,6 @@ export * from './retry-policy.ts'
|
||||
export { BlockAssembler } from './assembler.ts'
|
||||
export { callConfigEquals, deepFreeze, isAgentLoopRequest, markAgentLoopRequest } from './call-config.ts'
|
||||
export type { LlmCallConfig, LlmCallConfigAdapterDefaults } from './call-config.ts'
|
||||
export { isLlmAdapterFailure, llmFailureOf, llmRetryPolicyOf } from './adapter-failure.ts'
|
||||
|
||||
declare module 'cordis' {
|
||||
interface Context {
|
||||
@@ -124,10 +126,47 @@ export class LlmError extends HarnessError {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Accept one supplied credential, or refuse it as unusable.
|
||||
*
|
||||
* A stored key arrives from the credentials seam, a `.env` line, or a shell
|
||||
* export, all of which pick up surrounding whitespace, so trimming is silent.
|
||||
* Anything else fails here rather than inside `fetch`, whose ByteString
|
||||
* refusal names a UTF-16 code point instead of the setting to change. The key
|
||||
* never enters the message: `ref` names where to fix it, and echoing any part
|
||||
* of a secret into a log or a UI is the failure this diagnosis avoids.
|
||||
*
|
||||
* Lives beside {@link LlmError} rather than in `./api-key.ts` so the predicate
|
||||
* module stays dependency-free; both adapters share this one diagnosis instead
|
||||
* of keeping near-identical local copies.
|
||||
* @param raw - the credential exactly as supplied.
|
||||
* @param pkg - the refusing package name, prefixed to the diagnostic.
|
||||
* @param ref - the credential reference the value resolved through.
|
||||
* @returns the trimmed, usable key.
|
||||
*/
|
||||
export function assertUsableApiKey(raw: string, pkg: string, ref: string): string {
|
||||
const checked = normalizeApiKey(raw)
|
||||
if (checked.ok) return checked.value
|
||||
// The Models page is named as the writer it usually is, not as the only one:
|
||||
// the same value can arrive from a hand-edited .env or a shell export in a
|
||||
// composition that mounts no credentials seam at all, where directing the
|
||||
// user to a page that deployment does not serve would be a dead end.
|
||||
throw new LlmError(
|
||||
checked.reason === 'empty'
|
||||
? `${pkg}: the API key resolved from ${ref} is blank; set ${ref} to the raw key`
|
||||
+ ' (the web Models page writes it) or export it in the launching environment'
|
||||
: `${pkg}: the API key resolved from ${ref} contains characters no HTTP header can carry;`
|
||||
+ ` set ${ref} to the raw key alone (the web Models page writes it)`,
|
||||
INVALID_CREDENTIAL_CODE,
|
||||
)
|
||||
}
|
||||
|
||||
/** One model call whose config and adapter registration were resolved together. */
|
||||
export interface PreparedLlmCall {
|
||||
/** Detached, deep-frozen config with any adapter-owned default materialized. */
|
||||
readonly config: LlmCallConfig
|
||||
/** Immutable retry policy captured with the adapter registration. */
|
||||
readonly retryPolicy: ResolvedRetryPolicy
|
||||
/** Detached context metadata resolved with the registration-bound call. */
|
||||
readonly context?: LlmModelContext
|
||||
/** Config fields materialized by the captured adapter rather than proposed by the caller. */
|
||||
@@ -227,6 +266,27 @@ export interface AdapterRegistrationHandle {
|
||||
replace(providers: string[]): void
|
||||
}
|
||||
|
||||
/**
|
||||
* A live configurable-provider registration, disposable and atomically
|
||||
* replaceable — the directory counterpart of {@link AdapterRegistrationHandle}.
|
||||
*/
|
||||
export interface DirectoryRegistrationHandle {
|
||||
/** Withdraw every entry this registration currently holds. */
|
||||
(): void
|
||||
/**
|
||||
* Replace this registration's entries with `entries`. The candidate set is
|
||||
* validated in full first — an entry another registration already declares,
|
||||
* a duplicate within the set, or invalid metadata throws and leaves the
|
||||
* current entries untouched — and the swap is one synchronous section, so no
|
||||
* reader observes a gap. An empty array is legal here, unlike an empty
|
||||
* initial registration.
|
||||
*
|
||||
* Throws `LlmError` with code `REGISTRATION_DISPOSED` once the registration
|
||||
* has been disposed.
|
||||
*/
|
||||
replace(entries: readonly LlmConfigurableProvider[]): void
|
||||
}
|
||||
|
||||
/**
|
||||
* The abstract `llm` service: an adapter registry plus a streaming model-call
|
||||
* surface, interceptable via the `llm/stream` waterfall.
|
||||
@@ -234,6 +294,10 @@ export interface AdapterRegistrationHandle {
|
||||
export class LlmService extends Service {
|
||||
private adapters = new Map<string, AdapterRegistration>()
|
||||
private directory = new Map<string, LlmConfigurableProvider>()
|
||||
private discoveries = new Map<
|
||||
string,
|
||||
(request: LlmModelDiscoveryRequest) => Promise<readonly LlmDiscoveredModel[]>
|
||||
>()
|
||||
|
||||
constructor(ctx: Context) {
|
||||
super(ctx, 'llm')
|
||||
@@ -372,34 +436,61 @@ export class LlmService extends Service {
|
||||
* entry, or a provider already declared by any registration throws
|
||||
* `LlmError` without registering the rest. Disposed with the fiber.
|
||||
* @param entries - every configurable provider this plugin owns.
|
||||
* @returns the disposer that withdraws all of them.
|
||||
* @returns a handle that withdraws all of them, and can atomically replace them.
|
||||
*/
|
||||
registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void {
|
||||
const dispose = this.ctx.effect(function* (this: LlmService) {
|
||||
if (entries.length === 0) {
|
||||
throw new LlmError('a configurable-provider registration must declare at least one provider', 'INVALID_DIRECTORY')
|
||||
}
|
||||
registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle {
|
||||
let held: LlmConfigurableProvider[] = []
|
||||
let disposed = false
|
||||
/**
|
||||
* Validate a candidate set in full against everything this registration
|
||||
* does not already hold, then publish it. Nothing is written until the
|
||||
* whole set passes, so a refused candidate leaves the current entries in
|
||||
* place — the property that makes `replace` a swap rather than a
|
||||
* delete-then-add that can strand the directory empty.
|
||||
*/
|
||||
const commit = (candidates: readonly LlmConfigurableProvider[]): void => {
|
||||
const detached: LlmConfigurableProvider[] = []
|
||||
for (const entry of entries) {
|
||||
const own = new Set(held.map(entry => entry.provider))
|
||||
for (const entry of candidates) {
|
||||
if (entry.provider.length === 0 || entry.displayName.length === 0 || entry.settingsNs.length === 0) {
|
||||
throw new LlmError('configurable providers need a non-empty provider, displayName, and settingsNs', 'INVALID_DIRECTORY')
|
||||
}
|
||||
if (entry.settingsPath.some(segment => segment.length === 0)) {
|
||||
throw new LlmError(`configurable provider "${entry.provider}" has an empty settingsPath segment`, 'INVALID_DIRECTORY')
|
||||
}
|
||||
if (this.directory.has(entry.provider) || detached.some(seen => seen.provider === entry.provider)) {
|
||||
if ((this.directory.has(entry.provider) && !own.has(entry.provider))
|
||||
|| detached.some(seen => seen.provider === entry.provider)) {
|
||||
throw new LlmError(`configurable provider "${entry.provider}" is already declared`, 'DUPLICATE_DIRECTORY')
|
||||
}
|
||||
detached.push({ ...entry, settingsPath: [...entry.settingsPath] })
|
||||
}
|
||||
for (const entry of held) this.directory.delete(entry.provider)
|
||||
for (const entry of detached) this.directory.set(entry.provider, entry)
|
||||
held = detached
|
||||
this.emitAdaptersUpdated()
|
||||
}
|
||||
|
||||
const dispose = this.ctx.effect(function* (this: LlmService) {
|
||||
if (entries.length === 0) {
|
||||
throw new LlmError('a configurable-provider registration must declare at least one provider', 'INVALID_DIRECTORY')
|
||||
}
|
||||
commit(entries)
|
||||
yield () => {
|
||||
for (const entry of detached) this.directory.delete(entry.provider)
|
||||
disposed = true
|
||||
for (const entry of held) this.directory.delete(entry.provider)
|
||||
held = []
|
||||
this.emitAdaptersUpdated()
|
||||
}
|
||||
}.bind(this), 'llm.registerConfigurableProviders()')
|
||||
return () => void dispose()
|
||||
|
||||
const handle = ((): void => void dispose()) as DirectoryRegistrationHandle
|
||||
handle.replace = (next: readonly LlmConfigurableProvider[]): void => {
|
||||
if (disposed) {
|
||||
throw new LlmError('this configurable-provider registration was disposed', 'REGISTRATION_DISPOSED')
|
||||
}
|
||||
commit(next)
|
||||
}
|
||||
return handle
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -410,6 +501,73 @@ export class LlmService extends Service {
|
||||
return [...this.directory.values()].map(entry => ({ ...entry, settingsPath: [...entry.settingsPath] }))
|
||||
}
|
||||
|
||||
/**
|
||||
* Offer to interrogate provider endpoints on behalf of the settings
|
||||
* namespace this plugin owns. The namespace is the key because that is what
|
||||
* a configuration surface already holds from the configurable-provider
|
||||
* directory, and because a provider being *added* has no route to name yet.
|
||||
* Disposed with the fiber.
|
||||
* @param settingsNs - the namespace whose profiles this discovery serves.
|
||||
* @param discover - interrogates one endpoint; must honor `request.signal`.
|
||||
* @returns the disposer that withdraws the offer.
|
||||
*/
|
||||
registerModelDiscovery(
|
||||
settingsNs: string,
|
||||
discover: (request: LlmModelDiscoveryRequest) => Promise<readonly LlmDiscoveredModel[]>,
|
||||
): () => void {
|
||||
const dispose = this.ctx.effect(function* (this: LlmService) {
|
||||
if (settingsNs.length === 0) {
|
||||
throw new LlmError('model discovery needs a non-empty settings namespace', 'INVALID_DISCOVERY')
|
||||
}
|
||||
if (this.discoveries.has(settingsNs)) {
|
||||
throw new LlmError(`model discovery for "${settingsNs}" is already registered`, 'DUPLICATE_DISCOVERY')
|
||||
}
|
||||
this.discoveries.set(settingsNs, discover)
|
||||
yield () => {
|
||||
this.discoveries.delete(settingsNs)
|
||||
}
|
||||
}.bind(this), 'llm.registerModelDiscovery()')
|
||||
return () => void dispose()
|
||||
}
|
||||
|
||||
/**
|
||||
* Interrogate one provider endpoint for the models it advertises. The
|
||||
* request describes a draft, not a stored route, so nothing here reads or
|
||||
* writes settings or credentials — the caller owns both, and the reply is
|
||||
* candidate metadata a surface may offer for adoption.
|
||||
* @param settingsNs - namespace whose registered discovery serves this draft.
|
||||
* @param request - the endpoint, protocol, and one-shot credential to use.
|
||||
* @returns the advertised models, deduplicated in endpoint order.
|
||||
*/
|
||||
async discoverModels(
|
||||
settingsNs: string,
|
||||
request: LlmModelDiscoveryRequest,
|
||||
): Promise<LlmDiscoveredModel[]> {
|
||||
const discover = this.discoveries.get(settingsNs)
|
||||
if (discover === undefined) {
|
||||
throw new LlmError(`no model discovery is registered for "${settingsNs}"`, 'NO_DISCOVERY')
|
||||
}
|
||||
// One of the two identifies what to describe: a route the adapter knows, or
|
||||
// an endpoint to ask. Neither leaves nothing to answer about.
|
||||
if ((request.provider ?? '').length === 0 && (request.baseURL ?? '').length === 0) {
|
||||
throw new LlmError('model discovery needs a provider route or a baseURL', 'INVALID_DISCOVERY')
|
||||
}
|
||||
const discovered = await discover(request)
|
||||
const seen = new Set<string>()
|
||||
const models: LlmDiscoveredModel[] = []
|
||||
for (const model of discovered) {
|
||||
if (typeof model.id !== 'string' || model.id.length === 0 || seen.has(model.id)) continue
|
||||
seen.add(model.id)
|
||||
models.push({
|
||||
id: model.id,
|
||||
...model.name === undefined ? {} : { name: model.name },
|
||||
...model.contextWindow === undefined ? {} : { contextWindow: model.contextWindow },
|
||||
...model.maxTokens === undefined ? {} : { maxTokens: model.maxTokens },
|
||||
})
|
||||
}
|
||||
return models
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the retry policy captured when one provider route was registered.
|
||||
* @param provider - registered provider route to inspect.
|
||||
@@ -646,12 +804,19 @@ export class LlmService extends Service {
|
||||
let dispatched = false
|
||||
return Object.freeze({
|
||||
config: resolvedConfig,
|
||||
retryPolicy: registration.retryPolicy,
|
||||
adapterDefaults,
|
||||
...context === undefined ? {} : { context },
|
||||
stream: (options: GenerateOptions): AsyncIterable<StreamChunk> => {
|
||||
if (dispatched) {
|
||||
throw new LlmError('a prepared LLM call can only be dispatched once', 'INVALID_PREPARED_CALL')
|
||||
}
|
||||
if (!callConfigEquals(options, resolvedConfig)) {
|
||||
throw new LlmError(
|
||||
'prepared LLM call config changed before adapter dispatch',
|
||||
'INVALID_PREPARED_CALL',
|
||||
)
|
||||
}
|
||||
dispatched = true
|
||||
return this.streamWithRegistration(options, { registration, config: resolvedConfig })
|
||||
},
|
||||
@@ -681,22 +846,17 @@ export class LlmService extends Service {
|
||||
}
|
||||
|
||||
/**
|
||||
* Final adapter boundary. It tags only failures from adapter selection,
|
||||
* synchronous dispatch, iterator construction, or iteration while preserving
|
||||
* the original Error object. Middleware outside this generator remains
|
||||
* distinguishable as plugin work. An iteration failure skips adapter cleanup
|
||||
* so it cannot suppress the primary provider error. A downstream close awaits
|
||||
* adapter cleanup, whose failures remain ordinary untagged work.
|
||||
* Final adapter boundary. Adapter selection, dispatch, iterator construction,
|
||||
* and iteration failures become one terminal failure chunk. Middleware and
|
||||
* downstream consumer failures remain thrown plugin or consumer errors.
|
||||
*/
|
||||
private async * adapterStream(
|
||||
options: GenerateOptions,
|
||||
failures: AdapterFailureScope,
|
||||
prepared?: { registration: AdapterRegistration; config: LlmCallConfig },
|
||||
): AsyncGenerator<StreamChunk> {
|
||||
let iterator: AsyncIterator<StreamChunk>
|
||||
try {
|
||||
const registration = prepared?.registration ?? this.registration(options.provider)
|
||||
failures.retryPolicy = registration.retryPolicy
|
||||
const resolvedConfig = prepared === undefined
|
||||
? (await this.resolveCallFor(registration, options, options.signal)).config
|
||||
: prepared.config
|
||||
@@ -715,32 +875,34 @@ export class LlmService extends Service {
|
||||
const stream = adapter.stream(this.forAdapter(resolvedOptions, adapter))
|
||||
iterator = stream[Symbol.asyncIterator]()
|
||||
} catch (error: unknown) {
|
||||
throw markLlmAdapterFailure(failures, error)
|
||||
yield adapterFailureChunk(error, options.signal)
|
||||
return
|
||||
}
|
||||
|
||||
let completed = false
|
||||
let iterationFailed = false
|
||||
try {
|
||||
while (true) {
|
||||
let value: StreamChunk
|
||||
let item: { done: true } | { done: false; value: StreamChunk }
|
||||
try {
|
||||
const item = await iterator.next()
|
||||
if (item.done) {
|
||||
completed = true
|
||||
return
|
||||
}
|
||||
value = item.value
|
||||
const next = await iterator.next()
|
||||
item = next.done
|
||||
? { done: true }
|
||||
: { done: false, value: next.value }
|
||||
} catch (error: unknown) {
|
||||
iterationFailed = true
|
||||
throw markLlmAdapterFailure(failures, error)
|
||||
completed = true
|
||||
yield adapterFailureChunk(error, options.signal)
|
||||
return
|
||||
}
|
||||
if (item.done) {
|
||||
completed = true
|
||||
return
|
||||
}
|
||||
// End the adapter-owned try before yielding: consumer/middleware
|
||||
// failures resumed into this generator must remain untagged.
|
||||
yield value
|
||||
// failures resumed into this generator must remain thrown.
|
||||
yield item.value
|
||||
}
|
||||
} finally {
|
||||
// oxlint-disable-next-line typescript/no-unnecessary-condition -- the iteration catch sets its latch before entering finally.
|
||||
if (!completed && !iterationFailed) {
|
||||
if (!completed) {
|
||||
const close = iterator.return?.bind(iterator)
|
||||
if (close) await close()
|
||||
}
|
||||
@@ -748,15 +910,13 @@ export class LlmService extends Service {
|
||||
}
|
||||
|
||||
/**
|
||||
* Stream one model call as raw chunks (token-level deltas). Throws
|
||||
* `LlmError` with code `NO_ADAPTER` if no adapter is registered for
|
||||
* `options.provider`. Replay state is retained only when the same adapter
|
||||
* instance owns its historical provider and the target provider. Final
|
||||
* adapter selection remains fixed through asynchronous exact-model resolution
|
||||
* and dispatch. Selection, dispatch, and iteration failures retain their
|
||||
* original Error identity and are tagged in a call-local scope for narrow
|
||||
* agent-loop request recovery; middleware and nested-call failures remain
|
||||
* untagged for the outer call.
|
||||
* Stream one model call as raw chunks (token-level deltas). Replay state is
|
||||
* retained only when the same adapter instance owns its historical provider
|
||||
* and the target provider. Final adapter selection remains fixed through
|
||||
* asynchronous exact-model resolution and dispatch. Adapter selection,
|
||||
* dispatch, and iteration failures become terminal `error` or `aborted`
|
||||
* finish chunks; middleware, nested-call, cleanup, and consumer failures
|
||||
* remain thrown.
|
||||
* @param options - the full request; `options.provider` selects the adapter.
|
||||
* @returns the chunk stream, possibly wrapped by `llm/stream` listeners.
|
||||
*/
|
||||
@@ -768,14 +928,23 @@ export class LlmService extends Service {
|
||||
options: GenerateOptions,
|
||||
prepared?: { registration: AdapterRegistration; config: LlmCallConfig },
|
||||
): AsyncIterable<StreamChunk> {
|
||||
const failures: AdapterFailureScope = { failures: new WeakMap<Error, LlmFailure>() }
|
||||
const stream = this.ctx.waterfall(
|
||||
return this.ctx.waterfall(
|
||||
this,
|
||||
'llm/stream',
|
||||
options,
|
||||
() => this.adapterStream(options, failures, prepared),
|
||||
() => this.adapterStream(options, prepared),
|
||||
)
|
||||
return bindAdapterFailureScope(stream, failures)
|
||||
}
|
||||
}
|
||||
|
||||
/** Convert one adapter throw into the stream protocol's terminal outcome. */
|
||||
function adapterFailureChunk(error: unknown, signal?: AbortSignal): StreamChunk {
|
||||
const failure = normalizeLlmFailure(error)
|
||||
return {
|
||||
type: 'finish',
|
||||
reason: signal?.aborted || failure.code === 'ABORTED'
|
||||
? { kind: 'aborted', failure }
|
||||
: { kind: 'error', failure },
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -72,7 +72,9 @@ async function* validateStream(
|
||||
usageSeen = true
|
||||
break
|
||||
case 'finish':
|
||||
if (open.size > 0) fail(`LLM stream finished with ${open.size} open block(s)`)
|
||||
if (open.size > 0 && chunk.reason.kind !== 'error' && chunk.reason.kind !== 'aborted') {
|
||||
fail(`LLM stream finished with ${open.size} open block(s)`)
|
||||
}
|
||||
finished = true
|
||||
break
|
||||
}
|
||||
|
||||
@@ -29,17 +29,99 @@ export interface ToolMessageSource {
|
||||
callId: CallId
|
||||
}
|
||||
|
||||
/**
|
||||
* What SHAPE of information a producer-supplied context carries, declared by
|
||||
* the producer beside its provenance.
|
||||
*
|
||||
* `MessageSource.kind` answers *who produced this*; `form` answers *what kind
|
||||
* of thing it is*, and the two axes are deliberately independent — several
|
||||
* producers share one form (three snapshot producers today), and one producer
|
||||
* may emit more than one form over a session.
|
||||
*
|
||||
* The vocabulary is SEMANTIC, never visual: a value states that the content is
|
||||
* a file's instructions or a catalog of available items, and a consumer decides
|
||||
* what that looks like. Colors, icons, ordering, and collapse defaults are the
|
||||
* consumer's business and must not enter this union. It grows one value at a
|
||||
* time as producers gain the structured fields their form needs; an absent or
|
||||
* unknown value is the documented default, presented as opaque content.
|
||||
*/
|
||||
export type ContextForm =
|
||||
/** Instructions read out of workspace files the model is expected to follow. */
|
||||
| 'instructions'
|
||||
/** A catalog of items available in this session, republished as it changes. */
|
||||
| 'catalog'
|
||||
/** Current state, where a later snapshot from the same producer supersedes an earlier one. */
|
||||
| 'snapshot'
|
||||
/** A one-off account of something that just happened; it supersedes nothing. */
|
||||
| 'notice'
|
||||
/** A message another agent addressed to this one. */
|
||||
| 'relay'
|
||||
/** Material lifted out of another session's log, possibly reduced on the way in. */
|
||||
| 'recall'
|
||||
|
||||
/** One named contribution to a `snapshot`-form context, in assembly order. */
|
||||
export interface ContextSnapshotSection {
|
||||
/** The contributing subsystem's name. */
|
||||
readonly name: string
|
||||
/** That contribution's model-facing text, exactly as assembled. */
|
||||
readonly text: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Producer-declared {@link ContextForm} and the fields that form requires,
|
||||
* mixed into the source shapes that carry one.
|
||||
*
|
||||
* Discriminated by `form` so a producer cannot declare a shape without the
|
||||
* facts that shape is presented from: a `notice` must record its one-line
|
||||
* account, a `snapshot` its sections. Omitting `form` stays valid — an
|
||||
* undeclared context is the documented default.
|
||||
*/
|
||||
export type ContextFormed =
|
||||
| { readonly form?: never }
|
||||
| { readonly form: 'instructions' }
|
||||
| { readonly form: 'catalog' }
|
||||
| {
|
||||
readonly form: 'snapshot'
|
||||
/** The named contributions this snapshot assembled, in order. */
|
||||
readonly sections: readonly ContextSnapshotSection[]
|
||||
}
|
||||
| {
|
||||
readonly form: 'notice'
|
||||
/** One-line account of what happened, shown without expanding the row. */
|
||||
readonly summary: string
|
||||
}
|
||||
| { readonly form: 'relay' }
|
||||
| { readonly form: 'recall' }
|
||||
|
||||
/**
|
||||
* Where a message (or injected content) came from.
|
||||
* Merge-extensible sum type — plugins add their own `kind`s.
|
||||
*/
|
||||
export interface MessageSourceMap {
|
||||
user: { kind: 'user' }
|
||||
plugin: { kind: 'plugin'; plugin: string }
|
||||
plugin: { kind: 'plugin'; plugin: string } & ContextFormed
|
||||
model: ModelMessageSource
|
||||
tool: ToolMessageSource
|
||||
}
|
||||
|
||||
/**
|
||||
* Bound for a `notice` summary. The account rides a collapsed transcript row
|
||||
* and is committed to the durable log, while its inputs — task labels, goal
|
||||
* objectives, tool arguments — are caller text with no length of their own.
|
||||
*/
|
||||
export const CONTEXT_SUMMARY_MAX_CHARS = 120
|
||||
|
||||
/**
|
||||
* Bound one `notice` summary to {@link CONTEXT_SUMMARY_MAX_CHARS}.
|
||||
* @param summary - the producer's one-line account, of any length.
|
||||
* @returns the account, ellipsized when it exceeds the bound.
|
||||
*/
|
||||
export function boundContextSummary(summary: string): string {
|
||||
return summary.length <= CONTEXT_SUMMARY_MAX_CHARS
|
||||
? summary
|
||||
: `${summary.slice(0, CONTEXT_SUMMARY_MAX_CHARS - 1)}…`
|
||||
}
|
||||
|
||||
/** Any known message source, derived from {@link MessageSourceMap}; switch on `kind` and fall through unknowns (merge-extensible). */
|
||||
export type MessageSource = MessageSourceMap[keyof MessageSourceMap]
|
||||
|
||||
|
||||
@@ -160,6 +160,58 @@ export interface LlmConfigurableProvider {
|
||||
* object; empty when the whole section is the profile.
|
||||
*/
|
||||
settingsPath: readonly string[]
|
||||
/**
|
||||
* Whether the owning adapter knows this route only because configuration
|
||||
* declared it — a gateway or self-hosted server it ships nothing about.
|
||||
* Absent means the adapter draws no such distinction; false means it does
|
||||
* and this route is one of its own. Only the adapter can answer: a stored
|
||||
* profile is how a user-added route AND a corrected shipped one both look
|
||||
* from outside.
|
||||
*/
|
||||
declared?: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* One interrogation of a provider endpoint that configuration has not stored
|
||||
* yet. Configuration surfaces send the draft a user is still editing, so the
|
||||
* request carries the endpoint and credential directly instead of naming a
|
||||
* route: a provider being added has no route to name.
|
||||
*/
|
||||
export interface LlmModelDiscoveryRequest {
|
||||
/**
|
||||
* Route the draft is editing, when it edits an existing one. A route whose
|
||||
* adapter already knows its models answers from that knowledge instead of
|
||||
* asking the endpoint — the adapter's own registry is the better answer, and
|
||||
* it costs no network call.
|
||||
*/
|
||||
provider?: string
|
||||
/**
|
||||
* Endpoint to interrogate. Optional because a route the adapter already
|
||||
* describes needs none; a route it does not must supply one.
|
||||
*/
|
||||
baseURL?: string
|
||||
/** Wire protocol the endpoint speaks, when the draft names one. */
|
||||
api?: string
|
||||
/** Credential for this interrogation alone; the harness never stores it. */
|
||||
apiKey?: string
|
||||
/** Caller cancellation; implementations must settle promptly after it aborts. */
|
||||
signal?: AbortSignal
|
||||
}
|
||||
|
||||
/**
|
||||
* One model an endpoint reports about itself. Every field but the id is
|
||||
* optional because most provider listings disclose an id and nothing else;
|
||||
* a surface adopting one of these still owes the capacities its adapter needs.
|
||||
*/
|
||||
export interface LlmDiscoveredModel {
|
||||
/** Model id the endpoint accepts. */
|
||||
id: string
|
||||
/** Human-readable name when the endpoint supplies one. */
|
||||
name?: string
|
||||
/** Maximum combined request and response context, when disclosed. */
|
||||
contextWindow?: number
|
||||
/** Maximum output tokens, when disclosed. */
|
||||
maxTokens?: number
|
||||
}
|
||||
|
||||
/** One adapter-discovered model; catalog membership is advisory, not request validation. */
|
||||
@@ -217,8 +269,9 @@ export interface LlmResolvedModelInfo extends LlmModelInfo {
|
||||
* Raw streaming protocol emitted by adapters.
|
||||
* Block indexes correlate interleaved deltas, and `block-end` carries the
|
||||
* assembled block. Adapters emit usage before the terminal finish and nothing
|
||||
* afterward; tool arguments remain raw JSON strings. Failures either throw or
|
||||
* end with `error`/`aborted`, and consumers must handle both paths.
|
||||
* afterward; tool arguments remain raw JSON strings. An adapter implementation
|
||||
* may throw, but `LlmService.stream()` normalizes that failure to a terminal
|
||||
* `error` or `aborted` finish before exposing it to consumers.
|
||||
*/
|
||||
export type StreamChunk =
|
||||
| { type: 'block-start'; index: number; blockType: ContentBlockType }
|
||||
|
||||
68
packages/llm/llm/tests/adapter-failure.spec.ts
Normal file
68
packages/llm/llm/tests/adapter-failure.spec.ts
Normal file
@@ -0,0 +1,68 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { normalizeLlmFailure } from '../src/adapter-failure.ts'
|
||||
|
||||
describe('adapter failure normalization', () => {
|
||||
it('contains hostile non-Error coercion', () => {
|
||||
const thrown = { [Symbol.toPrimitive]: () => { throw new Error('coercion failed') } }
|
||||
expect(normalizeLlmFailure(thrown)).toEqual({ message: 'LLM adapter failed', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('normalizes empty primitive throws and data descriptors without values', () => {
|
||||
expect(normalizeLlmFailure('')).toEqual({ message: 'LLM adapter failed', code: 'UNKNOWN' })
|
||||
expect(normalizeLlmFailure(null)).toEqual({ message: 'null', code: 'UNKNOWN' })
|
||||
|
||||
const error = new Error('provider failed')
|
||||
Object.defineProperty(error, 'failure', { get: () => ({ message: 'ignored', code: 'IGNORED' }) })
|
||||
Object.defineProperty(error, 'code', { get: () => 'IGNORED' })
|
||||
expect(normalizeLlmFailure(error)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
|
||||
const accessorCode = Object.assign(new Error('provider failed'), {
|
||||
failure: { message: 'provider failed', code: 'FOREIGN' },
|
||||
})
|
||||
Object.defineProperty(accessorCode, 'code', { get: () => 'FOREIGN' })
|
||||
expect(normalizeLlmFailure(accessorCode)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
|
||||
const primitiveFailure = Object.assign(new Error('provider failed'), {
|
||||
failure: null,
|
||||
code: 'FOREIGN',
|
||||
})
|
||||
expect(normalizeLlmFailure(primitiveFailure)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('contains hostile Error property reflection', () => {
|
||||
const withFailure = new Error('provider failed') as Error & { failure: unknown; code: string }
|
||||
withFailure.failure = { message: 'provider failed', code: 'FOREIGN' }
|
||||
withFailure.code = 'FOREIGN'
|
||||
const hostileCode = new Proxy(withFailure, {
|
||||
getOwnPropertyDescriptor(target, property) {
|
||||
if (property === 'code') throw new Error('code descriptor failed')
|
||||
return Reflect.getOwnPropertyDescriptor(target, property)
|
||||
},
|
||||
})
|
||||
expect(normalizeLlmFailure(hostileCode)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
|
||||
const hostileFailure = new Proxy(new Error('provider failed'), {
|
||||
getOwnPropertyDescriptor() { throw new Error('failure descriptor failed') },
|
||||
})
|
||||
expect(normalizeLlmFailure(hostileFailure)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('rejects malformed or accessor-backed failure snapshots', () => {
|
||||
const malformed = new Error('provider failed') as Error & { failure: unknown; code: string }
|
||||
malformed.failure = { message: 'provider failed', code: 'FOREIGN', requestId: '' }
|
||||
malformed.code = 'FOREIGN'
|
||||
expect(normalizeLlmFailure(malformed)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
|
||||
const accessorBacked = new Error('provider failed') as Error & { failure: unknown }
|
||||
accessorBacked.failure = Object.defineProperty({}, 'message', {
|
||||
get() { throw new Error('failure getter failed') },
|
||||
})
|
||||
expect(normalizeLlmFailure(accessorBacked)).toEqual({ message: 'provider failed', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('falls back when an Error message accessor throws', () => {
|
||||
const error = new Error('provider failed')
|
||||
Object.defineProperty(error, 'message', { get() { throw new Error('message getter failed') } })
|
||||
expect(normalizeLlmFailure(error)).toEqual({ message: 'LLM adapter failed', code: 'UNKNOWN' })
|
||||
})
|
||||
})
|
||||
70
packages/llm/llm/tests/api-key.spec.ts
Normal file
70
packages/llm/llm/tests/api-key.spec.ts
Normal file
@@ -0,0 +1,70 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { assertUsableApiKey, INVALID_CREDENTIAL_CODE, normalizeApiKey } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
describe('normalizeApiKey', () => {
|
||||
it('accepts a printable-ASCII key unchanged', () => {
|
||||
expect(normalizeApiKey('sk-0123456789abcdef')).toEqual({ ok: true, value: 'sk-0123456789abcdef' })
|
||||
})
|
||||
|
||||
it('trims surrounding whitespace before judging', () => {
|
||||
expect(normalizeApiKey(' sk-abc\t\n')).toEqual({ ok: true, value: 'sk-abc' })
|
||||
})
|
||||
|
||||
it.each([
|
||||
['an empty string', ''],
|
||||
['spaces only', ' '],
|
||||
['a tab only', '\t'],
|
||||
])('rejects %s as empty', (_label, raw) => {
|
||||
expect(normalizeApiKey(raw)).toEqual({ ok: false, reason: 'empty' })
|
||||
})
|
||||
|
||||
it.each([
|
||||
['an emoji', 'sk-\u{1F600}abc'],
|
||||
['CJK text', 'sk-你好'],
|
||||
['full-width punctuation', 'sk-abc,'],
|
||||
['an interior space', 'sk-abc def'],
|
||||
['a C0 control character', 'sk-abc\x01'],
|
||||
['a latin-1 character', 'sk-café'],
|
||||
])('rejects %s as illegal characters', (_label, raw) => {
|
||||
expect(normalizeApiKey(raw)).toEqual({ ok: false, reason: 'illegalCharacters' })
|
||||
})
|
||||
|
||||
it('accepts the printable-ASCII boundary characters', () => {
|
||||
expect(normalizeApiKey('!~')).toEqual({ ok: true, value: '!~' })
|
||||
})
|
||||
|
||||
it('publishes a code distinct from a missing credential', () => {
|
||||
expect(INVALID_CREDENTIAL_CODE).toBe('INVALID_CREDENTIAL')
|
||||
})
|
||||
})
|
||||
|
||||
describe('assertUsableApiKey', () => {
|
||||
it('returns the trimmed key when it is usable', () => {
|
||||
expect(assertUsableApiKey(' sk-abc ', 'llm-deepseek', 'DEEPSEEK_API_KEY')).toBe('sk-abc')
|
||||
})
|
||||
|
||||
it('refuses a blank stored credential, naming the reference', () => {
|
||||
expect(() => assertUsableApiKey(' ', 'llm-deepseek', 'DEEPSEEK_API_KEY'))
|
||||
.toThrow(/llm-deepseek: the API key resolved from DEEPSEEK_API_KEY is blank/)
|
||||
})
|
||||
|
||||
it('refuses an unusable stored credential with the invalid-credential code', () => {
|
||||
try {
|
||||
assertUsableApiKey('sk-\u{1F600}', 'llm-pi-ai', 'ACME_API_KEY')
|
||||
expect.fail('an illegal key must throw')
|
||||
} catch (error) {
|
||||
expect((error as { code: string }).code).toBe(INVALID_CREDENTIAL_CODE)
|
||||
expect((error as Error).message).toContain('llm-pi-ai')
|
||||
expect((error as Error).message).toContain('ACME_API_KEY')
|
||||
}
|
||||
})
|
||||
|
||||
it('never echoes the key it refuses', () => {
|
||||
try {
|
||||
assertUsableApiKey('sk-\u{1F600}supersecret', 'llm-deepseek', 'DEEPSEEK_API_KEY')
|
||||
expect.fail('an illegal key must throw')
|
||||
} catch (error) {
|
||||
expect((error as Error).message).not.toContain('supersecret')
|
||||
}
|
||||
})
|
||||
})
|
||||
@@ -6,11 +6,8 @@ import LlmService, {
|
||||
HarnessError,
|
||||
isContextWindowExceededError,
|
||||
isQuotaExceededError,
|
||||
isLlmAdapterFailure,
|
||||
LlmAdapter,
|
||||
LlmError,
|
||||
llmFailureOf,
|
||||
llmRetryPolicyOf,
|
||||
ProviderRequestId,
|
||||
ReasoningEffortId,
|
||||
resolveRetryPolicy,
|
||||
@@ -95,6 +92,12 @@ const SCRIPT: StreamChunk[] = [
|
||||
{ type: 'finish', reason: { kind: 'stop' } },
|
||||
]
|
||||
|
||||
async function collect(stream: AsyncIterable<StreamChunk>): Promise<StreamChunk[]> {
|
||||
const chunks: StreamChunk[] = []
|
||||
for await (const chunk of stream) chunks.push(chunk)
|
||||
return chunks
|
||||
}
|
||||
|
||||
describe('LlmService', () => {
|
||||
it('recognizes structured and model-capacity context-window overflow details', () => {
|
||||
expect(isContextWindowExceededError('context_length_exceeded maximum context length')).toBe(true)
|
||||
@@ -142,6 +145,8 @@ describe('LlmService', () => {
|
||||
|
||||
it('errorChain survives non-Error values, hostile coercion, and circular causes', () => {
|
||||
expect(errorChain('plain string')).toBe('plain string')
|
||||
expect(errorChain({ message: 'structured provider failure', code: 'SERVER' }))
|
||||
.toBe('structured provider failure')
|
||||
expect(errorChain({ toString: () => { throw new Error('hostile') } })).toBe('<unrenderable value>')
|
||||
const circular = new Error('outer')
|
||||
circular.cause = circular
|
||||
@@ -219,69 +224,61 @@ describe('LlmService', () => {
|
||||
)
|
||||
})
|
||||
|
||||
it('keeps the serving registration policy on an in-flight call after route replacement', async () => {
|
||||
it('keeps a prepared registration and retry policy after route replacement', async () => {
|
||||
const oldPolicy = resolveRetryPolicy({ mode: 'always' }, 'old retryPolicy')
|
||||
const newPolicy = resolveRetryPolicy({ mode: 'normal', maxRetries: 0 }, 'new retryPolicy')
|
||||
const entered = Promise.withResolvers<undefined>()
|
||||
const release = Promise.withResolvers<undefined>()
|
||||
const failure = new LlmError('old route failed', 'AUTH')
|
||||
const oldAdapter = new class extends LlmAdapter {
|
||||
const oldFailure = new LlmError('old route failed', 'AUTH')
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const disposeOld = ctx.llm.registerAdapter(['route'], new class extends ThrowingAdapter {
|
||||
override providerRetryPolicy(): typeof oldPolicy {
|
||||
return oldPolicy
|
||||
}
|
||||
}(oldFailure))
|
||||
const prepared = await ctx.llm.prepareCall({ provider: 'route', model: 'model' })
|
||||
|
||||
async * stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
entered.resolve(undefined)
|
||||
await release.promise
|
||||
throw failure
|
||||
}
|
||||
}()
|
||||
const newAdapter = new class extends ScriptedAdapter {
|
||||
disposeOld()
|
||||
ctx.llm.registerAdapter(['route'], new class extends ScriptedAdapter {
|
||||
override providerRetryPolicy(): typeof newPolicy {
|
||||
return newPolicy
|
||||
}
|
||||
}(SCRIPT)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const disposeOld = ctx.llm.registerAdapter(['route'], oldAdapter)
|
||||
const stream = ctx.llm.stream({ provider: 'route', model: 'model', messages: [] })
|
||||
const outcome = (async (): Promise<unknown> => {
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
return error
|
||||
}
|
||||
return undefined
|
||||
})()
|
||||
await entered.promise
|
||||
}(SCRIPT))
|
||||
|
||||
disposeOld()
|
||||
ctx.llm.registerAdapter(['route'], newAdapter)
|
||||
release.resolve(undefined)
|
||||
|
||||
expect(await outcome).toBe(failure)
|
||||
expect(llmRetryPolicyOf(stream)).toBe(oldPolicy)
|
||||
const chunks = await collect(prepared.stream({ ...prepared.config, messages: [] }))
|
||||
expect(chunks.at(-1)).toEqual({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: { message: 'old route failed', code: 'AUTH' },
|
||||
},
|
||||
})
|
||||
expect(prepared.retryPolicy).toBe(oldPolicy)
|
||||
expect(ctx.llm.providerRetryPolicy('route')).toBe(newPolicy)
|
||||
})
|
||||
|
||||
it('throws NO_ADAPTER for unregistered providers', async () => {
|
||||
it('normalizes an unregistered provider to a terminal failure', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const stream = ctx.llm.stream({ provider: 'nope', model: 'any-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _ of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
expect(caught).toBeInstanceOf(LlmError)
|
||||
expect((caught as LlmError).code).toBe('NO_ADAPTER')
|
||||
expect((caught as LlmError).message).toContain('no adapter registered')
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(true)
|
||||
expect(llmRetryPolicyOf(stream)).toBeUndefined()
|
||||
|
||||
const chunks = await collect(ctx.llm.stream({
|
||||
provider: 'nope',
|
||||
model: 'any-model',
|
||||
messages: [],
|
||||
}))
|
||||
|
||||
const finish = chunks.at(-1)
|
||||
expect(finish).toMatchObject({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: { code: 'NO_ADAPTER' },
|
||||
},
|
||||
})
|
||||
if (finish?.type !== 'finish' || finish.reason.kind !== 'error') throw new Error('expected error finish')
|
||||
expect(finish.reason.failure.message).toContain('no adapter registered')
|
||||
})
|
||||
|
||||
it.each(['done', 'value'] as const)('tags a throwing IteratorResult.%s getter without replacing its Error', async (field) => {
|
||||
it.each(['done', 'value'] as const)('normalizes a throwing IteratorResult.%s getter', async (field) => {
|
||||
const original = new LlmError(`${field} getter failed`, 'RESULT_GETTER_FAILED')
|
||||
const result = field === 'done' ? {} : { done: false }
|
||||
Object.defineProperty(result, field, { get: () => { throw original } })
|
||||
@@ -297,31 +294,30 @@ describe('LlmService', () => {
|
||||
})
|
||||
const adapter = new class extends LlmAdapter {
|
||||
stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
return {
|
||||
[Symbol.asyncIterator](): AsyncIterator<StreamChunk> {
|
||||
return iterator
|
||||
},
|
||||
}
|
||||
return { [Symbol.asyncIterator]: () => iterator }
|
||||
}
|
||||
}()
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], adapter)
|
||||
ctx.llm.registerAdapter(['test'], adapter)
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
const chunks = await collect(ctx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
}))
|
||||
|
||||
expect(caught).toBe(original)
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(true)
|
||||
expect(chunks.at(-1)).toEqual({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: { message: `${field} getter failed`, code: 'RESULT_GETTER_FAILED' },
|
||||
},
|
||||
})
|
||||
expect(cleanupLookups).toBe(0)
|
||||
})
|
||||
|
||||
it.each(['dispatch', 'iterator'] as const)('tags synchronous adapter %s failures without replacing their Error', async (boundary) => {
|
||||
it.each(['dispatch', 'iterator'] as const)('normalizes synchronous adapter %s failures', async (boundary) => {
|
||||
const original = new LlmError(`${boundary} failed`, 'BOUNDARY_FAILED')
|
||||
const adapter = new class extends LlmAdapter {
|
||||
stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
@@ -331,339 +327,63 @@ describe('LlmService', () => {
|
||||
}()
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], adapter)
|
||||
ctx.llm.registerAdapter(['test'], adapter)
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
const chunks = await collect(ctx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
}))
|
||||
|
||||
expect(caught).toBe(original)
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(true)
|
||||
expect(llmFailureOf(stream, caught)).toEqual({
|
||||
message: `${boundary} failed`,
|
||||
code: 'BOUNDARY_FAILED',
|
||||
expect(chunks.at(-1)).toEqual({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: { message: `${boundary} failed`, code: 'BOUNDARY_FAILED' },
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps structured provider facts beside a frozen third-party Error', async () => {
|
||||
const original = new LlmError('provider busy', 'RATE_LIMIT', {
|
||||
it('preserves structured LlmError facts in the terminal failure', async () => {
|
||||
const failure = new LlmError('provider busy', 'RATE_LIMIT', {
|
||||
status: 429,
|
||||
providerRetryAfterMs: 1_500,
|
||||
requestId: ProviderRequestId('req-7'),
|
||||
})
|
||||
Object.freeze(original)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
ctx.llm.registerAdapter(['test'], new ThrowingAdapter(failure))
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
const chunks = await collect(ctx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
}))
|
||||
|
||||
expect(caught).toBe(original)
|
||||
expect(llmFailureOf(stream, caught)).toEqual({
|
||||
message: 'provider busy',
|
||||
code: 'RATE_LIMIT',
|
||||
status: 429,
|
||||
providerRetryAfterMs: 1_500,
|
||||
requestId: ProviderRequestId('req-7'),
|
||||
})
|
||||
})
|
||||
|
||||
it('does not trust retry facts carried by an unknown third-party Error', async () => {
|
||||
const carried = { message: 'busy', code: 'SERVER', status: 503 }
|
||||
const original = Object.assign(new Error('busy'), { failure: carried })
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
const facts = llmFailureOf(stream, original)
|
||||
carried.status = 500
|
||||
|
||||
expect(facts).toEqual({ message: 'busy', code: 'UNKNOWN' })
|
||||
expect(Object.isFrozen(facts)).toBe(true)
|
||||
expect(facts).not.toBe(carried)
|
||||
})
|
||||
|
||||
it('keeps validated failure facts across package copies with matching own codes', async () => {
|
||||
const original = Object.assign(new Error('provider busy'), {
|
||||
code: 'RATE_LIMIT',
|
||||
failure: {
|
||||
message: 'provider busy',
|
||||
code: 'RATE_LIMIT',
|
||||
status: 429,
|
||||
providerRetryAfterMs: 1_500,
|
||||
requestId: 'req-cross-copy',
|
||||
expect(chunks.at(-1)).toEqual({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: {
|
||||
message: 'provider busy',
|
||||
code: 'RATE_LIMIT',
|
||||
status: 429,
|
||||
providerRetryAfterMs: 1_500,
|
||||
requestId: ProviderRequestId('req-7'),
|
||||
},
|
||||
},
|
||||
})
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({
|
||||
message: 'provider busy',
|
||||
code: 'RATE_LIMIT',
|
||||
status: 429,
|
||||
providerRetryAfterMs: 1_500,
|
||||
requestId: 'req-cross-copy',
|
||||
})
|
||||
})
|
||||
|
||||
it('keeps an unknown SDK Error exact without trusting its private code or accessors', async () => {
|
||||
const original = Object.assign(new Error('socket closed'), { code: 'ECONNRESET' })
|
||||
Object.defineProperty(original, 'failure', {
|
||||
get() { throw new Error('SDK failure accessor must not run') },
|
||||
})
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
|
||||
expect(original.code).toBe('ECONNRESET')
|
||||
expect(llmFailureOf(stream, original)).toEqual({ message: 'socket closed', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('keeps an SDK Error exact when its message accessor is hostile', async () => {
|
||||
const original = Object.defineProperty(new Error(), 'message', {
|
||||
get() { throw new Error('SDK message accessor trap') },
|
||||
})
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({ message: 'LLM adapter failed', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('keeps an SDK Error exact without trusting accessor-backed carried facts', async () => {
|
||||
const original = Object.assign(new Error('busy'), {
|
||||
failure: { message: 'busy', code: 'SERVER', status: 503 },
|
||||
})
|
||||
Object.defineProperty(original, 'code', {
|
||||
get() { throw new Error('SDK code accessor must not escape') },
|
||||
})
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({ message: 'busy', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('does not trust carried facts matched only by an inherited code', async () => {
|
||||
class InheritedCodeError extends Error {
|
||||
get code(): string { return 'SERVER' }
|
||||
}
|
||||
const original = Object.assign(new InheritedCodeError('busy'), {
|
||||
failure: { message: 'busy', code: 'SERVER', status: 503 },
|
||||
})
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({ message: 'busy', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('keeps an SDK Error exact when code descriptor inspection is trapped', async () => {
|
||||
const target = Object.assign(new Error('busy'), {
|
||||
code: 'SERVER',
|
||||
failure: { message: 'busy', code: 'SERVER', status: 503 },
|
||||
})
|
||||
const original = new Proxy(target, {
|
||||
getOwnPropertyDescriptor(value, property) {
|
||||
if (property === 'code') throw new Error('SDK code descriptor trap')
|
||||
return Reflect.getOwnPropertyDescriptor(value, property)
|
||||
},
|
||||
})
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({ message: 'busy', code: 'UNKNOWN' })
|
||||
})
|
||||
|
||||
it('falls back safely when SDK objects trap failure inspection or expose malformed facts', async () => {
|
||||
const propertyTrap = new Proxy(new HarnessError('descriptor trapped', 'SERVER'), {
|
||||
getOwnPropertyDescriptor(target, property) {
|
||||
if (property === 'failure') throw new Error('SDK descriptor trap')
|
||||
return Reflect.getOwnPropertyDescriptor(target, property)
|
||||
},
|
||||
})
|
||||
const throwingFacts = Object.create(null) as Record<string, unknown>
|
||||
Object.defineProperty(throwingFacts, 'message', {
|
||||
get() { throw new Error('SDK fact getter trap') },
|
||||
})
|
||||
const carrying = (message: string, failure: unknown): HarnessError => Object.defineProperty(
|
||||
new HarnessError(message, 'SERVER'),
|
||||
'failure',
|
||||
{ value: failure },
|
||||
)
|
||||
const factGetter = carrying('fact getter failed', throwingFacts)
|
||||
const malformed = carrying('malformed facts', { message: 'provider busy', code: 'SERVER', requestId: 1 })
|
||||
const primitive = carrying('primitive facts', 1)
|
||||
const nullFacts = carrying('null facts', null)
|
||||
const mismatched = carrying('mismatched facts', { message: 'busy', code: 'RATE_LIMIT' })
|
||||
|
||||
for (const [original, expectedMessage] of [
|
||||
[propertyTrap, 'descriptor trapped'],
|
||||
[factGetter, 'fact getter failed'],
|
||||
[malformed, 'malformed facts'],
|
||||
[primitive, 'primitive facts'],
|
||||
[nullFacts, 'null facts'],
|
||||
[mismatched, 'mismatched facts'],
|
||||
] as const) {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({ message: expectedMessage, code: 'SERVER' })
|
||||
}
|
||||
})
|
||||
|
||||
it('retains a stable code from a HarnessError without requiring LlmError facts', async () => {
|
||||
const original = new HarnessError('stable adapter failure', 'ADAPTER_STABLE')
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-provider'], new ThrowingAdapter(original))
|
||||
const stream = ctx.llm.stream({ provider: 'test-provider', model: 'test-model', messages: [] })
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toBe(original)
|
||||
expect(llmFailureOf(stream, original)).toEqual({
|
||||
message: 'stable adapter failure',
|
||||
code: 'ADAPTER_STABLE',
|
||||
})
|
||||
expect(llmFailureOf(stream, 'not an Error')).toBeUndefined()
|
||||
expect(llmFailureOf({ [Symbol.asyncIterator]: () => stream[Symbol.asyncIterator]() }, original)).toBeUndefined()
|
||||
})
|
||||
|
||||
it('keeps a nested adapter failure scoped to the nested model call', async () => {
|
||||
const original = new LlmError('nested provider failed', 'NESTED_FAILED')
|
||||
const outer = new RecordingAdapter(SCRIPT)
|
||||
const nested = new ThrowingAdapter(original)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['outer'], outer)
|
||||
ctx.llm.registerAdapter(['nested'], nested)
|
||||
let nestedStream: AsyncIterable<StreamChunk> | undefined
|
||||
ctx.on('llm/stream', (options, next) => {
|
||||
if (options.provider !== 'outer') return next()
|
||||
return (async function* () {
|
||||
nestedStream = ctx.llm.stream({ provider: 'nested', model: 'nested', messages: [] })
|
||||
yield * nestedStream
|
||||
})()
|
||||
})
|
||||
|
||||
const outerStream = ctx.llm.stream({ provider: 'outer', model: 'outer', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of outerStream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
|
||||
expect(caught).toBe(original)
|
||||
expect(nestedStream).toBeDefined()
|
||||
expect(isLlmAdapterFailure(nestedStream!, caught)).toBe(true)
|
||||
expect(isLlmAdapterFailure(outerStream, caught)).toBe(false)
|
||||
expect(outer.lastOptions).toBeUndefined()
|
||||
})
|
||||
|
||||
it('keeps call scopes distinct when middleware reuses an iterable', async () => {
|
||||
const firstFailure = new LlmError('first provider failed', 'FIRST_FAILED')
|
||||
const secondFailure = new LlmError('second provider failed', 'SECOND_FAILED')
|
||||
const delegates: AsyncIterable<StreamChunk>[] = []
|
||||
const shared: AsyncIterable<StreamChunk> = {
|
||||
[Symbol.asyncIterator](): AsyncIterator<StreamChunk> {
|
||||
const delegate = delegates.shift()
|
||||
if (delegate === undefined) throw new Error('shared stream has no call delegate')
|
||||
return delegate[Symbol.asyncIterator]()
|
||||
},
|
||||
}
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['first'], new ThrowingAdapter(firstFailure))
|
||||
ctx.llm.registerAdapter(['second'], new ThrowingAdapter(secondFailure))
|
||||
ctx.on('llm/stream', (_options, next) => {
|
||||
delegates.push(next())
|
||||
return shared
|
||||
})
|
||||
|
||||
const firstStream = ctx.llm.stream({ provider: 'first', model: 'first', messages: [] })
|
||||
const secondStream = ctx.llm.stream({ provider: 'second', model: 'second', messages: [] })
|
||||
const catchFailure = async (stream: AsyncIterable<StreamChunk>): Promise<unknown> => {
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
return error
|
||||
}
|
||||
return new Error('expected adapter to fail')
|
||||
}
|
||||
|
||||
expect(firstStream).not.toBe(secondStream)
|
||||
const firstCaught = await catchFailure(firstStream)
|
||||
expect(firstCaught).toBe(firstFailure)
|
||||
expect(isLlmAdapterFailure(firstStream, firstCaught)).toBe(true)
|
||||
expect(isLlmAdapterFailure(secondStream, firstCaught)).toBe(false)
|
||||
const secondCaught = await catchFailure(secondStream)
|
||||
expect(secondCaught).toBe(secondFailure)
|
||||
expect(isLlmAdapterFailure(secondStream, secondCaught)).toBe(true)
|
||||
expect(isLlmAdapterFailure(firstStream, secondCaught)).toBe(false)
|
||||
expect(delegates).toHaveLength(0)
|
||||
})
|
||||
|
||||
it('propagates a rejected next promptly without awaiting a non-settling return', async () => {
|
||||
const original = new LlmError('provider failed', 'PROVIDER_FAILED')
|
||||
let cleanupCalls = 0
|
||||
it('normalizes arbitrary adapter rejections without throwing them downstream', async () => {
|
||||
const adapter = new class extends LlmAdapter {
|
||||
stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
return {
|
||||
[Symbol.asyncIterator](): AsyncIterator<StreamChunk> {
|
||||
return {
|
||||
next: () => Promise.reject(original),
|
||||
return: () => {
|
||||
cleanupCalls += 1
|
||||
return new Promise<IteratorResult<StreamChunk>>(() => {})
|
||||
},
|
||||
// Third-party adapters can reject with arbitrary values.
|
||||
// oxlint-disable-next-line typescript/prefer-promise-reject-errors
|
||||
next: () => Promise.reject('plain provider failure'),
|
||||
}
|
||||
},
|
||||
}
|
||||
@@ -671,30 +391,73 @@ describe('LlmService', () => {
|
||||
}()
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], adapter)
|
||||
ctx.llm.registerAdapter(['test'], adapter)
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })
|
||||
const failure = (async (): Promise<unknown> => {
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
return error
|
||||
}
|
||||
return new Error('expected adapter iteration to fail')
|
||||
})()
|
||||
let timer: ReturnType<typeof setTimeout> | undefined
|
||||
const timeout = new Promise<Error>((resolve) => {
|
||||
timer = setTimeout(() => { resolve(new Error('adapter failure did not settle promptly')) }, 100)
|
||||
const chunks = await collect(ctx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
}))
|
||||
|
||||
expect(chunks.at(-1)).toEqual({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: { message: 'plain provider failure', code: 'UNKNOWN' },
|
||||
},
|
||||
})
|
||||
const caught = await Promise.race([failure, timeout])
|
||||
if (timer !== undefined) clearTimeout(timer)
|
||||
|
||||
expect(caught).toBe(original)
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(true)
|
||||
expect(cleanupCalls).toBe(0)
|
||||
})
|
||||
|
||||
it('awaits one adapter return on downstream close and leaves its rejection unclassified', async () => {
|
||||
it('maps adapter failure to aborted when the request signal is aborted', async () => {
|
||||
const controller = new AbortController()
|
||||
controller.abort('cancelled')
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test'], new ThrowingAdapter(new Error('stopped')))
|
||||
|
||||
const chunks = await collect(ctx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
signal: controller.signal,
|
||||
}))
|
||||
|
||||
expect(chunks.at(-1)).toMatchObject({
|
||||
type: 'finish',
|
||||
reason: { kind: 'aborted', failure: { message: 'stopped' } },
|
||||
})
|
||||
})
|
||||
|
||||
it('leaves middleware and consumer failures thrown', async () => {
|
||||
const middlewareFailure = new Error('middleware failed')
|
||||
const middlewareCtx = new Context()
|
||||
await middlewareCtx.plugin(LlmService)
|
||||
middlewareCtx.llm.registerAdapter(['test'], new ScriptedAdapter(SCRIPT))
|
||||
middlewareCtx.on('llm/stream', () => (async function* () {
|
||||
throw middlewareFailure
|
||||
})())
|
||||
await expect(collect(middlewareCtx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
}))).rejects.toBe(middlewareFailure)
|
||||
|
||||
const consumerFailure = new Error('consumer failed')
|
||||
const consumerCtx = new Context()
|
||||
await consumerCtx.plugin(LlmService)
|
||||
consumerCtx.llm.registerAdapter(['test'], new ScriptedAdapter(SCRIPT))
|
||||
await expect((async () => {
|
||||
for await (const _chunk of consumerCtx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
})) {
|
||||
throw consumerFailure
|
||||
}
|
||||
})()).rejects.toBe(consumerFailure)
|
||||
})
|
||||
|
||||
it('awaits adapter cleanup on downstream close and leaves cleanup failure thrown', async () => {
|
||||
const cleanup = new Error('cleanup failed')
|
||||
let cleanupCalls = 0
|
||||
const adapter = new class extends LlmAdapter {
|
||||
@@ -714,22 +477,19 @@ describe('LlmService', () => {
|
||||
}()
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], adapter)
|
||||
ctx.llm.registerAdapter(['test'], adapter)
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of stream) break
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
|
||||
expect(caught).toBe(cleanup)
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(false)
|
||||
await expect((async () => {
|
||||
for await (const _chunk of ctx.llm.stream({
|
||||
provider: 'test',
|
||||
model: 'test',
|
||||
messages: [],
|
||||
})) break
|
||||
})()).rejects.toBe(cleanup)
|
||||
expect(cleanupCalls).toBe(1)
|
||||
})
|
||||
|
||||
it('allows downstream close when the adapter iterator has no return method', async () => {
|
||||
it('allows downstream close when an adapter iterator has no return method', async () => {
|
||||
const adapter = new class extends LlmAdapter {
|
||||
stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
return {
|
||||
@@ -741,66 +501,9 @@ describe('LlmService', () => {
|
||||
}()
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], adapter)
|
||||
ctx.llm.registerAdapter(['test'], adapter)
|
||||
|
||||
let chunks = 0
|
||||
for await (const _chunk of ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })) {
|
||||
chunks += 1
|
||||
break
|
||||
}
|
||||
|
||||
expect(chunks).toBe(1)
|
||||
})
|
||||
|
||||
it('normalizes and tags non-Error adapter failures once', async () => {
|
||||
const adapter = new class extends LlmAdapter {
|
||||
stream(_options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
return {
|
||||
[Symbol.asyncIterator](): AsyncIterator<StreamChunk> {
|
||||
// Third-party adapters can reject with arbitrary values.
|
||||
// oxlint-disable-next-line typescript/prefer-promise-reject-errors
|
||||
return { next: () => Promise.reject('plain provider failure') }
|
||||
},
|
||||
}
|
||||
}
|
||||
}()
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], adapter)
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
|
||||
expect(caught).toBeInstanceOf(HarnessError)
|
||||
expect(caught).toMatchObject({ code: 'UNKNOWN', cause: 'plain provider failure' })
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(true)
|
||||
})
|
||||
|
||||
it('does not tag a failure thrown downstream while consuming adapter output', async () => {
|
||||
const downstream = new Error('consumer failed')
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
ctx.llm.registerAdapter(['test-model'], new ScriptedAdapter(SCRIPT))
|
||||
|
||||
const stream = ctx.llm.stream({ provider: 'test-model', model: 'test-model', messages: [] })
|
||||
let caught: unknown
|
||||
try {
|
||||
for await (const _chunk of stream) throw downstream
|
||||
} catch (error: unknown) {
|
||||
caught = error
|
||||
}
|
||||
|
||||
expect(caught).toBe(downstream)
|
||||
expect(isLlmAdapterFailure(stream, caught)).toBe(false)
|
||||
expect(isLlmAdapterFailure(new ScriptedAdapter(SCRIPT).stream({
|
||||
provider: 'unbound', model: 'unbound', messages: [],
|
||||
}), caught)).toBe(false)
|
||||
expect(isLlmAdapterFailure(stream, 'consumer failed')).toBe(false)
|
||||
for await (const _chunk of ctx.llm.stream({ provider: 'test', model: 'test', messages: [] })) break
|
||||
})
|
||||
|
||||
it('unregisters adapters when the owning fiber is disposed (HMR safety)', async () => {
|
||||
@@ -1140,19 +843,34 @@ describe('LlmService', () => {
|
||||
expect(Object.isFrozen(prepared.config)).toBe(true)
|
||||
expect(Object.isFrozen(prepared.adapterDefaults)).toBe(true)
|
||||
expect(prepared.adapterDefaults).toEqual({ reasoningEffort: true })
|
||||
const stream = prepared.stream({
|
||||
expect(() => prepared.stream({
|
||||
...prepared.config,
|
||||
model: 'other',
|
||||
messages: [],
|
||||
})
|
||||
|
||||
await expect((async () => {
|
||||
for await (const _chunk of stream) { /* drain */ }
|
||||
})()).rejects.toMatchObject({ code: 'INVALID_PREPARED_CALL' })
|
||||
})).toThrow(expect.objectContaining({ code: 'INVALID_PREPARED_CALL' }))
|
||||
await collect(prepared.stream({
|
||||
...prepared.config,
|
||||
messages: [],
|
||||
}))
|
||||
expect(() => prepared.stream({
|
||||
...prepared.config,
|
||||
messages: [],
|
||||
})).toThrow(expect.objectContaining({ code: 'INVALID_PREPARED_CALL' }))
|
||||
|
||||
const late = await ctx.llm.prepareCall({ provider: 'route', model: 'model' })
|
||||
const lateOptions = { ...late.config, messages: [] }
|
||||
const lateStream = late.stream(lateOptions)
|
||||
lateOptions.model = 'other'
|
||||
expect(await collect(lateStream)).toContainEqual({
|
||||
type: 'finish',
|
||||
reason: {
|
||||
kind: 'error',
|
||||
failure: {
|
||||
message: 'prepared LLM call config changed before adapter dispatch',
|
||||
code: 'INVALID_PREPARED_CALL',
|
||||
},
|
||||
},
|
||||
})
|
||||
})
|
||||
|
||||
it('reuses one exact-model lookup for prepared config and context metadata', async () => {
|
||||
|
||||
@@ -170,6 +170,32 @@ describe('configurable-provider directory', () => {
|
||||
expect(ctx.llm.listConfigurableProviders()).toEqual([])
|
||||
})
|
||||
|
||||
it('replaces its entries atomically, keeping the old set when a candidate collides', async () => {
|
||||
const ctx = await setup()
|
||||
const handle = ctx.llm.registerConfigurableProviders([entry(), entry({ provider: 'second' })])
|
||||
ctx.llm.registerConfigurableProviders([entry({ provider: 'owned-elsewhere' })])
|
||||
|
||||
// A candidate another registration already declares refuses the whole swap.
|
||||
expect(() =>{ handle.replace([entry({ provider: 'owned-elsewhere' })]) }).toThrow(/already declared/)
|
||||
expect(ctx.llm.listConfigurableProviders().map(view => view.provider).sort())
|
||||
.toEqual(['owned-elsewhere', 'second', entry().provider].sort())
|
||||
|
||||
// Its own entries are not "already declared" against itself, so a swap that
|
||||
// keeps one and drops another lands whole.
|
||||
handle.replace([entry({ displayName: 'Renamed' })])
|
||||
expect(ctx.llm.listConfigurableProviders().map(view => view.provider).sort())
|
||||
.toEqual(['owned-elsewhere', entry().provider].sort())
|
||||
expect(ctx.llm.listConfigurableProviders().find(view => view.provider === entry().provider)?.displayName)
|
||||
.toBe('Renamed')
|
||||
|
||||
// An empty replace is legal, unlike an empty initial registration.
|
||||
handle.replace([])
|
||||
expect(ctx.llm.listConfigurableProviders().map(view => view.provider)).toEqual(['owned-elsewhere'])
|
||||
|
||||
handle()
|
||||
expect(() =>{ handle.replace([entry()]) }).toThrow(/was disposed/)
|
||||
})
|
||||
|
||||
it('rejects duplicates within one registration and across registrations', async () => {
|
||||
const ctx = await setup()
|
||||
expect(() => ctx.llm.registerConfigurableProviders([entry(), entry()])).toThrow(/already declared/)
|
||||
@@ -179,3 +205,64 @@ describe('configurable-provider directory', () => {
|
||||
expect(ctx.llm.listConfigurableProviders()).toHaveLength(1)
|
||||
})
|
||||
})
|
||||
|
||||
describe('model discovery registry', () => {
|
||||
it('offers one interrogation per settings namespace and disposes with its fiber', async () => {
|
||||
const ctx = await setup()
|
||||
const discover = vi.fn(() => Promise.resolve([{ id: 'from-endpoint' }]))
|
||||
|
||||
const dispose = ctx.llm.registerModelDiscovery('llm-example', discover)
|
||||
await expect(ctx.llm.discoverModels('llm-example', { baseURL: 'https://gateway.example/v1' }))
|
||||
.resolves.toEqual([{ id: 'from-endpoint' }])
|
||||
expect(discover).toHaveBeenCalledWith({ baseURL: 'https://gateway.example/v1' })
|
||||
|
||||
// Disposal is observed through the offer itself, which is the only thing
|
||||
// the registration ever produced.
|
||||
dispose()
|
||||
await expect(ctx.llm.discoverModels('llm-example', { baseURL: 'https://gateway.example/v1' }))
|
||||
.rejects.toThrow(/no model discovery is registered/)
|
||||
})
|
||||
|
||||
it('rejects an unnamed namespace and a second registration of the same one', async () => {
|
||||
const ctx = await setup()
|
||||
const discover = (): Promise<never[]> => Promise.resolve([])
|
||||
|
||||
expect(() => ctx.llm.registerModelDiscovery('', discover)).toThrow(/non-empty settings namespace/)
|
||||
ctx.llm.registerModelDiscovery('llm-example', discover)
|
||||
expect(() => ctx.llm.registerModelDiscovery('llm-example', discover)).toThrow(/already registered/)
|
||||
// The refused second registration left the first one serving.
|
||||
await expect(ctx.llm.discoverModels('llm-example', { baseURL: 'https://gateway.example/v1' }))
|
||||
.resolves.toEqual([])
|
||||
})
|
||||
|
||||
it('normalizes what an interrogation returns without inventing capacities', async () => {
|
||||
const ctx = await setup()
|
||||
ctx.llm.registerModelDiscovery('llm-example', () => Promise.resolve([
|
||||
{ id: 'keep', name: 'Keep', contextWindow: 1024, maxTokens: 256 },
|
||||
{ id: '' },
|
||||
{ id: 'keep' },
|
||||
{ id: 'bare' },
|
||||
] as never))
|
||||
|
||||
expect(await ctx.llm.discoverModels('llm-example', { baseURL: 'https://gateway.example/v1' })).toEqual([
|
||||
{ id: 'keep', name: 'Keep', contextWindow: 1024, maxTokens: 256 },
|
||||
{ id: 'bare' },
|
||||
])
|
||||
})
|
||||
|
||||
it('refuses a namespace nothing serves and a draft with no endpoint', async () => {
|
||||
const ctx = await setup()
|
||||
ctx.llm.registerModelDiscovery('llm-example', () => Promise.resolve([]))
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-absent', { baseURL: 'https://gateway.example/v1' }))
|
||||
.rejects.toMatchObject({ code: 'NO_DISCOVERY' })
|
||||
await expect(ctx.llm.discoverModels('llm-example', { baseURL: '' }))
|
||||
.rejects.toMatchObject({ code: 'INVALID_DISCOVERY' })
|
||||
await expect(ctx.llm.discoverModels('llm-example', { provider: '', baseURL: '' }))
|
||||
.rejects.toMatchObject({ code: 'INVALID_DISCOVERY' })
|
||||
await expect(ctx.llm.discoverModels('llm-example', {}))
|
||||
.rejects.toMatchObject({ code: 'INVALID_DISCOVERY' })
|
||||
// Naming a route alone is enough: the adapter may know it without an endpoint.
|
||||
await expect(ctx.llm.discoverModels('llm-example', { provider: 'known-route' })).resolves.toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/token-meter/README.md
|
||||
README.md: 701893b342f9a93a75bec175634b1054f3d17151
|
||||
README.zh.md: a5844e8788422bba669632ed587fb87e1e2a1e58
|
||||
README.md: 8f868f25f3c4caf1fdab5b50965aab41efecf5af
|
||||
README.zh.md: 3621105ff35606b62b0587063038116b4772c6cf
|
||||
|
||||
@@ -23,19 +23,23 @@ Usage accounting sums disjoint input, cache-read, cache-write, and output bucket
|
||||
|
||||
## Session projections
|
||||
|
||||
When the composition provides `ctx.sessionProjections`, token-meter registers two units through an optional child fiber.
|
||||
When the composition provides `ctx.sessionProjections`, token-meter registers three units through an optional child fiber.
|
||||
|
||||
`tokenUsage` carries the complete durable log's `uncachedInputTokens`, `outputTokens`, `cacheReadTokens`, and `cacheWriteTokens`. Usage chunks are counted even when a request later fails; a final assistant-message usage for the same `(turn, step)` replaces that sample instead of double-counting it. Reasoning remains an output subdivision. The single last-sample slot relies on a session-log ordering property: once a later step reports usage, a legal log never reports usage for an earlier step again.
|
||||
|
||||
`contextPressure` carries optional `pressureTokens` — the newest provider-reported prompt size, summing uncached input plus cache reads and writes — and optional `contextWindow` from the newest `request/context` record. Pressure stays absent until a provider reports usage; capacity stays absent for a route whose adapter advertises none. Output is excluded, so the numerator holds still while a turn streams and steps forward when the next request reports its usage.
|
||||
`contextPressure` carries optional `pressureTokens` — the newest provider-reported prompt size, summing uncached input plus cache reads and writes — optional `projectedTokens`, and optional `contextWindow` from the newest `request/context` record. Both figures stay absent until a provider reports usage; capacity stays absent for a route whose adapter advertises none. Output is excluded, so `pressureTokens` holds still while a turn streams and steps forward when the next request reports its usage.
|
||||
|
||||
Both units use the standard projection baseline, live frame, higher-seq-wins store, and JSON checkpoint paths. Unloading token-meter removes both keys. A headless or TUI composition without the projection seam keeps the measurement service's existing behavior.
|
||||
`projectedTokens` is what the NEXT request's prompt would cost: the sample plus the heuristic repricing of everything the surface gained or lost since it was taken, clamped at zero and folded through the same `surface-fold.ts` the measurement service replays. Only the delta is estimated, so the figure stays anchored to the provider while reacting the moment content lands — or a compaction shadows a span. That last case is why the field exists: compaction summarizes through a direct `ctx.llm.stream()` call and appends no usage of its own, so `pressureTokens` alone reports the pre-compaction prompt until an entire further turn completes. Occupancy displays read `projectedTokens`.
|
||||
|
||||
`contextBreakdown` carries heuristic `systemTokens`, `toolsTokens`, and `messageTokens` — the context's composition rather than its provider-billed size. The envelope figures reprice last-wins on every `request/header`; the message figure replays `surface-fold.ts` — the same positional fold `measure()` runs — so it equals `measure().surfaceTokens` at every event boundary and compaction shrinks it the way it shrinks the next request. All three figures use the measurement service's fixed heuristic and are estimates: they will not sum to `projectedTokens`, whose provider anchor carries exactly the error — CJK text and JSON schemas underprice badly at four characters per token — that the composition rows still contain. Present them as an approximate composition, never as a total.
|
||||
|
||||
All three units use the standard projection baseline, live frame, higher-seq-wins store, and JSON checkpoint paths. Unloading token-meter removes all three keys. A composition without the projection seam keeps the measurement service's existing behavior.
|
||||
|
||||
### Context occupancy is an approximation, by design
|
||||
|
||||
`pressureTokens` and `contextWindow` are independent last-wins fields and are **not** one atomic observation of a single request. Switching models pairs the fresh capacity with the previous route's pressure until the next request reports usage, and `pressureTokens` describes the last request rather than the surface as it stands right now.
|
||||
The occupancy fields are independent last-wins records and are **not** one atomic observation of a single request. Switching models pairs the fresh capacity with the previous route's sample until the next request reports usage, and `pressureTokens` describes the last request rather than the surface as it stands right now — `projectedTokens` carries that sample forward over the surface's movement, but its anchor is still the older request.
|
||||
|
||||
This is deliberate. An occupancy percentage is a user-facing reference figure, not a billing record or a gating input — nothing in the harness makes decisions from it, and compaction reads `measure()` instead. The TUI status line has always computed occupancy the same way, dividing a `measure()` total by a separately-resolved capacity for the selected model.
|
||||
This is deliberate. An occupancy percentage is a user-facing reference figure, not a billing record or a gating input — nothing in the harness makes decisions from it, and compaction reads `measure()` instead. A UI computes occupancy by dividing measured pressure by the separately resolved capacity for the selected model.
|
||||
|
||||
Making the pair atomic was tried and rejected: it required a transient non-replayable wire frame, which needed lifecycle fencing against cross-stream reordering and left occupancy blank after every reconnect. The [Agent Note](../../../.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md) records that comparison. Consumers that need an exact same-boundary figure should call `measure()` at their own request boundary rather than read this projection.
|
||||
|
||||
@@ -62,4 +66,3 @@ No direct invalidation; the named consumer owns any request-prefix changes.
|
||||
- **Every measurement clones the current surface** — coherent immutable snapshots make reads O(surface), including below-threshold pressure checks.
|
||||
- **Provider usage is only reusable for an identical canonical envelope** — prompt, prefix, tools, provider, model, or call-config changes deliberately fall back to full heuristic estimation.
|
||||
- **Legacy provenance is conservative** — assistant messages without `sourceEventSeqs` cannot distinguish provider output from listener rewrites, so the fold avoids claiming a known empty or exact chunk stream.
|
||||
- **The TUI and browser fixture retain parallel folds** — `tokenUsage` owns durable session-projection semantics; the TUI keeps its live per-step map because its composition does not mount the generic projection seam, while the browser fixture mirrors the unit for standalone demo data.
|
||||
|
||||
@@ -23,21 +23,25 @@ fold 跟踪完整请求标头快照、步骤边界、表层追加与替换、成
|
||||
|
||||
## 会话投影
|
||||
|
||||
当组合提供 `ctx.sessionProjections` 时,token-meter 会通过一个可选子 fiber 注册两个单元。
|
||||
当组合提供 `ctx.sessionProjections` 时,token-meter 会通过一个可选子 fiber 注册三个单元。
|
||||
|
||||
`tokenUsage` 携带完整持久日志中的 `uncachedInputTokens`、`outputTokens`、`cacheReadTokens` 和 `cacheWriteTokens`。即使请求随后失败,用量分片仍会计入;同一 `(turn, step)` 的最终 assistant 消息用量会替换该样本,而不是重复计数。推理仍是输出的一个细分项。只保留单个最新样本,依赖的是会话日志的一条顺序性质:一旦某个更晚的步骤报告了用量,合法日志就绝不会再为更早的步骤报告用量。
|
||||
|
||||
`contextPressure` 携带可选的 `pressureTokens`(提供方报告的最新提示词规模,为未缓存输入加缓存读取与写入之和),以及来自最新一条 `request/context` 记录的可选 `contextWindow`。提供方报告用量前压力保持缺失;路由适配器未公布容量时容量也保持缺失。输出不计入其中,因此轮次流式输出期间分子保持不动,等到下一个请求报告用量时才前进。
|
||||
`contextPressure` 携带可选的 `pressureTokens`(提供方报告的最新提示词规模,为未缓存输入加缓存读取与写入之和)、可选的 `projectedTokens`,以及来自最新一条 `request/context` 记录的可选 `contextWindow`。提供方报告用量前两个数字都保持缺失;路由适配器未公布容量时容量也保持缺失。输出不计入其中,因此轮次流式输出期间 `pressureTokens` 保持不动,等到下一个请求报告用量时才前进。
|
||||
|
||||
两个单元都使用标准的投影基线、实时帧、seq 高者胜值仓和 JSON 检查点路径。卸载 token-meter 会移除这两个键。不带投影 seam 的 headless 或 TUI 组合会保留测量服务的既有行为。
|
||||
`projectedTokens` 是「下一个请求的提示词要花多少」:在该样本之上,加上自取样以来表层增减部分的启发式重新计价,下界钳制为零,折叠走的是测量服务重放的同一份 `surface-fold.ts`。只有增量部分是估算的,因此这个数字既锚定在提供方读数上,又能在内容落地——或压缩遮蔽一段区间——的瞬间做出反应。最后这种情况正是该字段存在的理由:压缩通过直连的 `ctx.llm.stream()` 调用生成摘要,自身不追加任何用量,所以仅凭 `pressureTokens` 会一直报告压缩前的提示词规模,直到又跑完一整轮为止。占用率展示读取 `projectedTokens`。
|
||||
|
||||
`contextBreakdown` 携带启发式的 `systemTokens`、`toolsTokens` 与 `messageTokens`,描述上下文的组成而非提供方计费规模。envelope 数字在每条 `request/header` 上按后者胜重新计价;消息数字重放 `surface-fold.ts`——与 `measure()` 运行的位置折叠是同一份——因此它在每个事件边界上都等于 `measure().surfaceTokens`,压缩会像缩小下一个请求那样缩小它。三个数字都使用测量服务的固定启发式规则,属于估算值:它们加起来不等于 `projectedTokens`——后者的提供方锚点恰好把这些明细行仍然带着的误差排除在外(按「4 字符 ≈ 1 token」计价,CJK 文本与 JSON schema 会被严重低估)。请把它们当作近似的**组成**呈现,而不是总量。
|
||||
|
||||
三个单元都使用标准的投影基线、实时帧、seq 高者胜值仓和 JSON 检查点路径。卸载 token-meter 会移除这三个键。不带投影 seam 的组合会保留测量服务的既有行为。
|
||||
|
||||
### 上下文占用率是刻意为之的近似值
|
||||
|
||||
`pressureTokens` 与 `contextWindow` 是两个各自后者胜的独立字段,**不是**对单个请求的一次原子观测。切换模型时,新容量会与上一路由的压力配对,直到下一个请求报告用量为止;而 `pressureTokens` 描述的是最后一个请求,不是此刻的表层。
|
||||
这些占用率字段各自后者胜、彼此独立,**不是**对单个请求的一次原子观测。切换模型时,新容量会与上一路由的样本配对,直到下一个请求报告用量为止;而 `pressureTokens` 描述的是最后一个请求,不是此刻的表层——`projectedTokens` 把该样本沿表层的增减推进到当下,但它的锚点仍然是那个较早的请求。
|
||||
|
||||
这是刻意的选择。占用率百分比是面向用户的参考数字,既不是计费记录,也不是门控输入:harness 中没有任何环节依据它做决策,压缩改为直接读取 `measure()`。TUI 状态行一直以同样的方式计算占用率,即用 `measure()` 总量除以为所选模型单独解析出的容量。
|
||||
这是刻意的选择。占用率百分比是面向用户的参考数字,既不是计费记录,也不是门控输入:harness 中没有任何环节依据它做决策,压缩改为直接读取 `measure()`。UI 用测得的压力除以为所选模型单独解析出的容量来计算占用率。
|
||||
|
||||
让这对值保持原子已经尝试过并被否决:它需要一个临时且不可回放的协议帧,进而需要针对跨流重排序的生命周期栅栏,还会让占用率在每次重连后变为空白。[Agent Note(agent 决策记录)](../../../.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md)记录了这项对比。需要同一边界精确数字的消费方应在自己的请求边界调用 `measure()`,而不是读取该投影。
|
||||
让这对值保持原子已经尝试过并被否决:它需要一个临时且不可回放的协议帧,进而需要针对跨流重排序的生命周期栅栏,还会让占用率在每次重连后变为空白。[Agent Note](../../../.agents/notes/implemented/architecture/2026-07-29-projected-token-usage-and-request-context.md)记录了这项对比。需要同一边界精确数字的消费方应在自己的请求边界调用 `measure()`,而不是读取该投影。
|
||||
|
||||
## 组合
|
||||
|
||||
@@ -62,4 +66,3 @@ fold 跟踪完整请求标头快照、步骤边界、表层追加与替换、成
|
||||
- **每次测量都会克隆当前表层**:一致且不可变的快照使读取成为 O(surface),包括低于阈值的压力检查。
|
||||
- **提供方用量只能为完全相同的规范 envelope 复用**:提示词、前缀、工具、提供方、模型或调用配置变更都会有意回退到完整启发式估算。
|
||||
- **遗留溯源采取保守策略**:没有 `sourceEventSeqs` 的 assistant 消息无法区分提供方输出与 listener 改写,因此 fold 不会声称已知空流或精确分片流。
|
||||
- **TUI 与浏览器 fixture 仍保留并行 fold**:`tokenUsage` 拥有持久会话投影语义;TUI 的组合未挂载通用投影 seam,因此继续维护实时的逐步骤 map,而浏览器 fixture 会为独立 demo 数据镜像该单元。
|
||||
|
||||
@@ -26,12 +26,11 @@
|
||||
"lib/index.js",
|
||||
"lib/invariant.js",
|
||||
"lib/types/**/*.js",
|
||||
"lib/types/**/*.d.ts",
|
||||
"lib/types/**/*.d.ts.map",
|
||||
"src"
|
||||
"lib/types/**/*.d.ts"
|
||||
],
|
||||
"license": "BSD-3-Clause",
|
||||
"peerDependencies": {
|
||||
"@deepseek-ai/dsh-compact": "^0.0.1",
|
||||
"@deepseek-ai/dsh-invariants": "^0.0.1",
|
||||
"@deepseek-ai/dsh-llm": "^0.0.1",
|
||||
"@deepseek-ai/dsh-session": "^0.0.1",
|
||||
@@ -43,6 +42,7 @@
|
||||
"zod": "^4.4.3"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@deepseek-ai/dsh-compact": "workspace:^",
|
||||
"@deepseek-ai/dsh-invariants": "workspace:^",
|
||||
"@deepseek-ai/dsh-llm": "workspace:^",
|
||||
"@deepseek-ai/dsh-session": "workspace:^",
|
||||
|
||||
70
packages/llm/token-meter/src/breakdown-projection.ts
Normal file
70
packages/llm/token-meter/src/breakdown-projection.ts
Normal file
@@ -0,0 +1,70 @@
|
||||
/**
|
||||
* Pure fold for the heuristic context-composition projection: system prompt
|
||||
* and tool schemas from the newest request envelope, conversation from the
|
||||
* live surface. Prices with the same shared estimator as the meter service,
|
||||
* so the three figures match `measure()`'s heuristic vocabulary exactly.
|
||||
*/
|
||||
|
||||
import { z } from 'zod'
|
||||
import { canonicalHeader } from '@deepseek-ai/dsh-session'
|
||||
import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection'
|
||||
import { estimateSystemTokens, estimateToolsTokens } from './estimate.ts'
|
||||
import { foldSurfaceProjection } from './surface-projection.ts'
|
||||
import type { ShadowPriceClaim } from './surface-projection.ts'
|
||||
// Import for the `contextBreakdown` SessionProjectionMap key merge.
|
||||
import type {} from './projection.ts'
|
||||
|
||||
interface ContextBreakdownState {
|
||||
systemTokens: number
|
||||
toolsTokens: number
|
||||
messageTokens: number
|
||||
/** Shadow price armed by the immediately preceding metering event. */
|
||||
claim?: ShadowPriceClaim
|
||||
}
|
||||
|
||||
const breakdownSchema = z.object({
|
||||
systemTokens: z.number().int().nonnegative(),
|
||||
toolsTokens: z.number().int().nonnegative(),
|
||||
messageTokens: z.number().int().nonnegative(),
|
||||
}).strict()
|
||||
|
||||
/**
|
||||
* Token-meter's context-composition projection unit.
|
||||
*
|
||||
* Envelope figures are last-wins per `request/header`; the message figure
|
||||
* rides {@link foldSurfaceProjection} — the same O(1) fold the occupancy
|
||||
* projection uses — so fully metered logs equal `measure().surfaceTokens` at
|
||||
* every event boundary and compaction shrinks the figure by its logged shadow
|
||||
* price. A replacement without a claim preserves the previous total. The
|
||||
* state is a fixed handful of numbers, so the persisted checkpoint stays
|
||||
* O(1) over the session's life.
|
||||
*/
|
||||
export const contextBreakdownProjectionDefinition:
|
||||
ProjectionDefinition<'contextBreakdown', ContextBreakdownState> = {
|
||||
key: 'contextBreakdown',
|
||||
schema: breakdownSchema,
|
||||
init: () => ({ systemTokens: 0, toolsTokens: 0, messageTokens: 0 }),
|
||||
apply: (state, event) => {
|
||||
const fold = foldSurfaceProjection(state.claim, event)
|
||||
let systemTokens = state.systemTokens
|
||||
let toolsTokens = state.toolsTokens
|
||||
if (event.type === 'request/header') {
|
||||
const header = canonicalHeader(event.data.header)
|
||||
systemTokens = estimateSystemTokens(header)
|
||||
toolsTokens = estimateToolsTokens(header)
|
||||
}
|
||||
if (systemTokens === state.systemTokens
|
||||
&& toolsTokens === state.toolsTokens
|
||||
&& fold.deltaTokens === 0
|
||||
&& fold.claim === undefined
|
||||
&& state.claim === undefined) return state
|
||||
return {
|
||||
systemTokens,
|
||||
toolsTokens,
|
||||
messageTokens: state.messageTokens + fold.deltaTokens,
|
||||
...fold.claim === undefined ? {} : { claim: fold.claim },
|
||||
}
|
||||
},
|
||||
view: ({ systemTokens, toolsTokens, messageTokens }) => ({ systemTokens, toolsTokens, messageTokens }),
|
||||
stateVersion: 2,
|
||||
}
|
||||
87
packages/llm/token-meter/src/estimate.ts
Normal file
87
packages/llm/token-meter/src/estimate.ts
Normal file
@@ -0,0 +1,87 @@
|
||||
/**
|
||||
* Fixed-density heuristic token pricing shared by the meter service and the
|
||||
* pure context-breakdown projection, so both surfaces price identical content
|
||||
* to identical numbers.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-token-meter/estimate
|
||||
*/
|
||||
|
||||
import type { ContentBlock, Message } from '@deepseek-ai/dsh-llm'
|
||||
import type { EpochHeader } from '@deepseek-ai/dsh-session'
|
||||
|
||||
/** Fixed text-density estimate used until exact tokenization is needed. */
|
||||
const CHARS_PER_TOKEN = 4
|
||||
|
||||
/** Per-block structural overhead for JSON framing and type tags. */
|
||||
const BLOCK_OVERHEAD = 4
|
||||
|
||||
/** Role-field framing overhead added to every priced message. */
|
||||
export const ROLE_OVERHEAD = 4
|
||||
|
||||
/**
|
||||
* Price content blocks recursively under the fixed density heuristic.
|
||||
* @param blocks - content blocks to price without mutation.
|
||||
* @returns heuristic tokens including per-block structural overhead.
|
||||
*/
|
||||
export function estimateContent(blocks: readonly ContentBlock[]): number {
|
||||
let tokens = 0
|
||||
for (const block of blocks) {
|
||||
switch (block.type) {
|
||||
case 'text':
|
||||
case 'reasoning':
|
||||
tokens += Math.ceil(block.text.length / CHARS_PER_TOKEN) + BLOCK_OVERHEAD
|
||||
break
|
||||
case 'tool-call':
|
||||
tokens += Math.ceil(block.name.length / CHARS_PER_TOKEN)
|
||||
+ Math.ceil(block.arguments.length / CHARS_PER_TOKEN)
|
||||
+ BLOCK_OVERHEAD
|
||||
break
|
||||
case 'tool-result':
|
||||
tokens += estimateContent(block.content) + BLOCK_OVERHEAD
|
||||
break
|
||||
default:
|
||||
// ContentBlockMap is merge-extensible; unknown blocks retain a
|
||||
// conservative structural JSON price under the fixed heuristic.
|
||||
tokens += BLOCK_OVERHEAD + Math.ceil(JSON.stringify(block).length / CHARS_PER_TOKEN)
|
||||
}
|
||||
}
|
||||
return tokens
|
||||
}
|
||||
|
||||
/**
|
||||
* Heuristically price one model-visible message.
|
||||
* @param message - message to price without mutation.
|
||||
* @returns content and role-framing tokens under the fixed heuristic.
|
||||
*/
|
||||
export function estimateMessage(message: Message): number {
|
||||
return estimateContent(message.content) + ROLE_OVERHEAD
|
||||
}
|
||||
|
||||
/**
|
||||
* Price the system-prompt part of a canonical request envelope.
|
||||
* @param header - canonical envelope, or undefined before any request.
|
||||
* @returns heuristic system-prompt tokens; 0 when absent.
|
||||
*/
|
||||
export function estimateSystemTokens(header: EpochHeader | undefined): number {
|
||||
if (header?.system === undefined) return 0
|
||||
return Math.ceil(header.system.length / CHARS_PER_TOKEN) + ROLE_OVERHEAD
|
||||
}
|
||||
|
||||
/**
|
||||
* Price the tool-schema part of a canonical request envelope.
|
||||
* @param header - canonical envelope, or undefined before any request.
|
||||
* @returns heuristic tool-schema tokens; 0 when absent or empty.
|
||||
*/
|
||||
export function estimateToolsTokens(header: EpochHeader | undefined): number {
|
||||
if (header?.tools === undefined || header.tools.length === 0) return 0
|
||||
return Math.ceil(JSON.stringify(header.tools).length / CHARS_PER_TOKEN) + BLOCK_OVERHEAD
|
||||
}
|
||||
|
||||
/**
|
||||
* Price the complete non-surface request envelope.
|
||||
* @param header - canonical envelope, or undefined before any request.
|
||||
* @returns heuristic system plus tool tokens.
|
||||
*/
|
||||
export function estimateHeader(header: EpochHeader | undefined): number {
|
||||
return estimateSystemTokens(header) + estimateToolsTokens(header)
|
||||
}
|
||||
@@ -7,8 +7,8 @@
|
||||
import { Context, Service } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import { BlockAssembler, deepFreeze } from '@deepseek-ai/dsh-llm'
|
||||
import type { ContentBlock, Message, TokenUsage } from '@deepseek-ai/dsh-llm'
|
||||
import type { EpochHeader, Session, SessionEvent, SurfaceEvent } from '@deepseek-ai/dsh-session'
|
||||
import type { Message, TokenUsage } from '@deepseek-ai/dsh-llm'
|
||||
import type { EpochHeader, Session, SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import { canonicalHeader, headerEquals, isSurfaceEvent } from '@deepseek-ai/dsh-session'
|
||||
// Type-only: resolves the optional projection registry Context seam.
|
||||
import type {} from '@deepseek-ai/dsh-session-projection'
|
||||
@@ -18,19 +18,13 @@ import type {
|
||||
TokenMeterConfig,
|
||||
TokenSurfaceNode,
|
||||
} from './types.ts'
|
||||
import { contextBreakdownProjectionDefinition } from './breakdown-projection.ts'
|
||||
import { contextPressureProjectionDefinition, tokenUsageProjectionDefinition } from './usage-projection.ts'
|
||||
import { estimateContent, estimateHeader, estimateMessage, ROLE_OVERHEAD } from './estimate.ts'
|
||||
import { foldSurfaceTokens } from './surface-fold.ts'
|
||||
|
||||
export type * from './types.ts'
|
||||
|
||||
/** Fixed text-density estimate used until exact tokenization is needed. */
|
||||
const CHARS_PER_TOKEN = 4
|
||||
|
||||
/** Per-block structural overhead for JSON framing and type tags. */
|
||||
const BLOCK_OVERHEAD = 4
|
||||
|
||||
/** Role-field framing overhead added to every priced message. */
|
||||
const ROLE_OVERHEAD = 4
|
||||
|
||||
interface MeasurementAnchor {
|
||||
readonly header: EpochHeader | undefined
|
||||
readonly surfaceTokens: number
|
||||
@@ -46,11 +40,6 @@ interface ReplayState {
|
||||
anchor: MeasurementAnchor | undefined
|
||||
}
|
||||
|
||||
interface PreparedSurfaceMutation {
|
||||
readonly tokens: number
|
||||
commit(state: ReplayState): void
|
||||
}
|
||||
|
||||
/** Sum disjoint provider usage buckets without double-counting reasoning output. */
|
||||
function usageTokens(usage: TokenUsage): number {
|
||||
return usage.inputTokens
|
||||
@@ -93,11 +82,12 @@ export class TokenMeterService extends Service {
|
||||
super(ctx, 'tokenMeter')
|
||||
validateConfigKeys(config)
|
||||
|
||||
// Projection registration is an optional child: headless and TUI
|
||||
// compositions without the generic registry keep the meter's old shape.
|
||||
// Projection registration is an optional child: compositions without the
|
||||
// generic registry keep the meter's standalone read shape.
|
||||
ctx.inject(['sessionProjections'], (projectionCtx) => {
|
||||
projectionCtx.sessionProjections.register(tokenUsageProjectionDefinition)
|
||||
projectionCtx.sessionProjections.register(contextPressureProjectionDefinition)
|
||||
projectionCtx.sessionProjections.register(contextBreakdownProjectionDefinition)
|
||||
})
|
||||
|
||||
// Readers catch up independently, while eager observation bounds ordinary
|
||||
@@ -141,7 +131,7 @@ export class TokenMeterService extends Service {
|
||||
} else {
|
||||
baseline = {
|
||||
kind: 'estimated',
|
||||
tokens: this._estimateHeader(header) + state.surfaceTokens,
|
||||
tokens: estimateHeader(header) + state.surfaceTokens,
|
||||
}
|
||||
surfaceDeltaTokens = 0
|
||||
}
|
||||
@@ -157,12 +147,13 @@ export class TokenMeterService extends Service {
|
||||
}
|
||||
|
||||
/**
|
||||
* Heuristically price one model-visible message.
|
||||
* Heuristically price one model-visible message (instance face of the pure
|
||||
* `estimateMessage` export from `estimate.ts`).
|
||||
* @param message - message to price without mutation.
|
||||
* @returns content and role-framing tokens under the fixed service heuristic.
|
||||
*/
|
||||
estimateMessage(message: Message): number {
|
||||
return this._estimateContent(message.content) + ROLE_OVERHEAD
|
||||
return estimateMessage(message)
|
||||
}
|
||||
|
||||
/** Catch one session's fold up to the current durable tail. */
|
||||
@@ -224,7 +215,7 @@ export class TokenMeterService extends Service {
|
||||
}
|
||||
|
||||
const surface = isSurfaceEvent(event)
|
||||
? this._prepareSurfaceMutation(session, state, event)
|
||||
? foldSurfaceTokens(state.surface, event)
|
||||
: undefined
|
||||
|
||||
if (event.type === 'assistant/message') {
|
||||
@@ -246,7 +237,7 @@ export class TokenMeterService extends Service {
|
||||
)
|
||||
const anchorSurfaceTokens = stepStart.surfaceTokens + providerAssistantTokens
|
||||
const providerTokens = usageTokens(event.data.usage)
|
||||
const estimatedAnchorTokens = this._estimateHeader(nextHeader) + anchorSurfaceTokens
|
||||
const estimatedAnchorTokens = estimateHeader(nextHeader) + anchorSurfaceTokens
|
||||
nextAnchor = {
|
||||
header: nextHeader,
|
||||
surfaceTokens: anchorSurfaceTokens,
|
||||
@@ -263,7 +254,7 @@ export class TokenMeterService extends Service {
|
||||
surfaceTokens: anchorSurfaceTokens,
|
||||
baseline: {
|
||||
kind: 'estimated',
|
||||
tokens: this._estimateHeader(nextHeader) + anchorSurfaceTokens,
|
||||
tokens: estimateHeader(nextHeader) + anchorSurfaceTokens,
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -271,53 +262,13 @@ export class TokenMeterService extends Service {
|
||||
|
||||
state.header = nextHeader
|
||||
state.stepStart = nextStepStart
|
||||
if (surface !== undefined) surface.commit(state)
|
||||
if (surface !== undefined) {
|
||||
state.surface = surface.nodes
|
||||
state.surfaceTokens += surface.deltaTokens
|
||||
}
|
||||
state.anchor = nextAnchor
|
||||
}
|
||||
|
||||
/** Validate one surface operation and return its allocation-light commit. */
|
||||
private _prepareSurfaceMutation(
|
||||
session: Session,
|
||||
state: ReplayState,
|
||||
event: SurfaceEvent,
|
||||
): PreparedSurfaceMutation {
|
||||
const tokens = this._estimateSurfaceEvent(session, event)
|
||||
const op = event.surfaceOp
|
||||
if (op === 'append') {
|
||||
return {
|
||||
tokens,
|
||||
commit(target) {
|
||||
target.surface.push({ seq: event.seq, tokens })
|
||||
target.surfaceTokens += tokens
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
const startIdx = state.surface.findIndex(node => node.seq === op.start)
|
||||
const endIdx = state.surface.findIndex(node => node.seq === op.end)
|
||||
if (startIdx === -1 || endIdx === -1 || startIdx > endIdx) {
|
||||
throw new Error(
|
||||
`token meter: replace at seq ${event.seq} has invalid current range ${op.start}-${op.end}`,
|
||||
)
|
||||
}
|
||||
const removedTokens = state.surface
|
||||
.slice(startIdx, endIdx + 1)
|
||||
.reduce((total, node) => total + node.tokens, 0)
|
||||
return {
|
||||
tokens,
|
||||
commit(target) {
|
||||
target.surface.splice(startIdx, endIdx - startIdx + 1, { seq: event.seq, tokens })
|
||||
target.surfaceTokens += tokens - removedTokens
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/** Price one current surface event exactly as it projects to a request. */
|
||||
private _estimateSurfaceEvent(session: Session, event: SurfaceEvent): number {
|
||||
const message = session.deriveEventMessage(event)
|
||||
return message === null ? 0 : this.estimateMessage(message)
|
||||
}
|
||||
|
||||
/**
|
||||
* Reassemble provider output from exact chunk provenance for a usage anchor.
|
||||
* Missing legacy provenance conservatively treats the durable output as the
|
||||
@@ -355,46 +306,7 @@ export class TokenMeterService extends Service {
|
||||
assembler.push(sourceEvent.data.chunk)
|
||||
}
|
||||
const providerContent = assembler.blocks()
|
||||
return providerContent.length === 0 ? 0 : this._estimateContent(providerContent) + ROLE_OVERHEAD
|
||||
}
|
||||
|
||||
/** Price content blocks recursively under the fixed density heuristic. */
|
||||
private _estimateContent(blocks: readonly ContentBlock[]): number {
|
||||
let tokens = 0
|
||||
for (const block of blocks) {
|
||||
switch (block.type) {
|
||||
case 'text':
|
||||
case 'reasoning':
|
||||
tokens += Math.ceil(block.text.length / CHARS_PER_TOKEN) + BLOCK_OVERHEAD
|
||||
break
|
||||
case 'tool-call':
|
||||
tokens += Math.ceil(block.name.length / CHARS_PER_TOKEN)
|
||||
+ Math.ceil(block.arguments.length / CHARS_PER_TOKEN)
|
||||
+ BLOCK_OVERHEAD
|
||||
break
|
||||
case 'tool-result':
|
||||
tokens += this._estimateContent(block.content) + BLOCK_OVERHEAD
|
||||
break
|
||||
default:
|
||||
// ContentBlockMap is merge-extensible; unknown blocks retain a
|
||||
// conservative structural JSON price under the fixed heuristic.
|
||||
tokens += BLOCK_OVERHEAD + Math.ceil(JSON.stringify(block).length / CHARS_PER_TOKEN)
|
||||
}
|
||||
}
|
||||
return tokens
|
||||
}
|
||||
|
||||
/** Price the canonical non-surface request envelope. */
|
||||
private _estimateHeader(header: EpochHeader | undefined): number {
|
||||
if (header === undefined) return 0
|
||||
let tokens = 0
|
||||
if (header.system !== undefined) {
|
||||
tokens += Math.ceil(header.system.length / CHARS_PER_TOKEN) + ROLE_OVERHEAD
|
||||
}
|
||||
if (header.tools !== undefined && header.tools.length > 0) {
|
||||
tokens += Math.ceil(JSON.stringify(header.tools).length / CHARS_PER_TOKEN) + BLOCK_OVERHEAD
|
||||
}
|
||||
return tokens
|
||||
return providerContent.length === 0 ? 0 : estimateContent(providerContent) + ROLE_OVERHEAD
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -17,9 +17,14 @@ export const inject = ['invariants']
|
||||
/**
|
||||
* No runtime invariant: token estimates are per-call outputs and the private
|
||||
* session cache is invalidated at its event mutation boundary. The package's
|
||||
* projection does expose an observation stream, but its schema fixes the JSON
|
||||
* payload and its pure fold replaces same-step samples; totals need not be
|
||||
* monotone when a final usage sample corrects an earlier chunk.
|
||||
* three projections do expose observation streams, but their schemas fix the
|
||||
* JSON payloads; the usage folds replace same-step samples, so totals need not
|
||||
* be monotone when a final sample corrects an earlier chunk, and the
|
||||
* composition fold prices through the same `estimate.ts` heuristic as the
|
||||
* measurement service and subtracts producer-logged shadow prices derived
|
||||
* from that service's own nodes, which makes its message figure equal
|
||||
* `measure().surfaceTokens` by construction rather than by a relation worth
|
||||
* observing at runtime.
|
||||
*/
|
||||
const install: InvariantInstaller = () => {}
|
||||
|
||||
|
||||
@@ -20,14 +20,12 @@ export interface TokenUsageProjection {
|
||||
/**
|
||||
* Approximate context occupancy for a status display.
|
||||
*
|
||||
* The two fields, when present, are deliberately NOT one atomic request
|
||||
* observation: `pressureTokens` is the newest provider-reported prompt size,
|
||||
* `contextWindow` the newest recorded route capacity. Switching models can
|
||||
* therefore pair a fresh capacity with the previous route's pressure until the
|
||||
* next request reports usage. This is an intentional trade — the value is a
|
||||
* user-facing reference, not a billing or gating input — and it matches how
|
||||
* the TUI status line has always computed occupancy. See the token-meter
|
||||
* README for the full rationale.
|
||||
* The fields, when present, are deliberately NOT one atomic request
|
||||
* observation: each is a last-wins record of a different moment. Switching
|
||||
* models can therefore pair a fresh capacity with the previous route's
|
||||
* pressure until the next request reports usage. This is an intentional trade
|
||||
* — the value is a user-facing reference, not a billing or gating input. See
|
||||
* the token-meter README for the full rationale.
|
||||
*/
|
||||
export interface ContextPressureProjection {
|
||||
/**
|
||||
@@ -36,15 +34,44 @@ export interface ContextPressureProjection {
|
||||
* grow as the current turn streams. Absent until a provider reports usage.
|
||||
*/
|
||||
pressureTokens?: number
|
||||
/**
|
||||
* What the NEXT request's prompt would cost: {@link pressureTokens} plus the
|
||||
* heuristic repricing of everything the surface gained or lost since that
|
||||
* sample. Only the delta is estimated, so the figure stays anchored to the
|
||||
* provider while still reacting the moment a compaction shadows a span —
|
||||
* which `pressureTokens` alone cannot do, since compaction reports no usage
|
||||
* of its own. Absent until a provider reports usage.
|
||||
*/
|
||||
projectedTokens?: number
|
||||
/** Newest recorded route capacity; absent when no adapter advertised one. */
|
||||
contextWindow?: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Heuristic composition of the next request's context: what the prompt is
|
||||
* made of, not what it costs. All three figures use the meter's fixed
|
||||
* density estimate, so they will not sum to the provider-anchored
|
||||
* `projectedTokens`: the estimator systematically underprices CJK text and
|
||||
* JSON schemas, which is exactly the error the anchoring in
|
||||
* {@link ContextPressureProjection.projectedTokens} keeps out of the occupancy
|
||||
* figure. Present these as approximations of composition, never as a total.
|
||||
*/
|
||||
export interface ContextBreakdownProjection {
|
||||
/** Heuristic tokens of the newest request envelope's system prompt; 0 before any request. */
|
||||
systemTokens: number
|
||||
/** Heuristic tokens of the newest request envelope's tool schemas; 0 before any request. */
|
||||
toolsTokens: number
|
||||
/** Heuristic tokens of the current model-visible conversation surface. */
|
||||
messageTokens: number
|
||||
}
|
||||
|
||||
declare module '@deepseek-ai/dsh-session-projection/types' {
|
||||
interface SessionProjectionMap {
|
||||
/** Provider-reported usage accumulated across the complete durable log. */
|
||||
tokenUsage: TokenUsageProjection
|
||||
/** Newest request pressure paired with the newest known route capacity. */
|
||||
contextPressure: ContextPressureProjection
|
||||
/** Heuristic system/tools/message composition of the next request. */
|
||||
contextBreakdown: ContextBreakdownProjection
|
||||
}
|
||||
}
|
||||
|
||||
65
packages/llm/token-meter/src/surface-fold.ts
Normal file
65
packages/llm/token-meter/src/surface-fold.ts
Normal file
@@ -0,0 +1,65 @@
|
||||
/**
|
||||
* The measurement service's positional surface fold: the per-node priced
|
||||
* surface `measure()` serves and compaction plans against. The projection
|
||||
* units deliberately do NOT share this fold — their state must stay O(1)
|
||||
* for the persisted checkpoint, so they ride `surface-projection.ts`'s
|
||||
* shadow-price protocol instead. Fully metered logs stay in agreement by
|
||||
* construction: both price through `estimate.ts`, and every logged shadow
|
||||
* price is derived from THIS fold's nodes by the replace producer. A
|
||||
* projection replacement without a claim deliberately folds with zero delta.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-token-meter/surface-fold
|
||||
*/
|
||||
|
||||
import { deriveEventMessage } from '@deepseek-ai/dsh-session'
|
||||
import type { SurfaceEvent } from '@deepseek-ai/dsh-session'
|
||||
import type { TokenSurfaceNode } from './types.ts'
|
||||
import { estimateMessage } from './estimate.ts'
|
||||
|
||||
/** One surface event's placement and cost against the surface preceding it. */
|
||||
export interface SurfaceTokenFold {
|
||||
/** Heuristic price of the event's own message; 0 when it derives none. */
|
||||
readonly tokens: number
|
||||
/** The surface after the event, detached from the input. */
|
||||
readonly nodes: TokenSurfaceNode[]
|
||||
/** Signed change in the surface total: `tokens` minus anything shadowed. */
|
||||
readonly deltaTokens: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Fold one surface event onto a priced surface.
|
||||
*
|
||||
* Total and allocation-fresh: the caller assigns the result rather than
|
||||
* mutating in place, so a throw here leaves the caller's state untouched and
|
||||
* the same malformed event fails identically on every retry.
|
||||
* @param nodes - the priced surface preceding this event, in model-visible order.
|
||||
* @param event - the surface event to place.
|
||||
* @returns the event's price, the next surface, and the signed total delta.
|
||||
* @throws when a replacement names a range absent from `nodes` — committed
|
||||
* logs are surface-validated at append time, so an unresolvable range is log
|
||||
* corruption and must fail loud rather than skip the event.
|
||||
*/
|
||||
export function foldSurfaceTokens(
|
||||
nodes: readonly TokenSurfaceNode[],
|
||||
event: SurfaceEvent,
|
||||
): SurfaceTokenFold {
|
||||
const message = deriveEventMessage(event)
|
||||
const tokens = message === null ? 0 : estimateMessage(message)
|
||||
const op = event.surfaceOp
|
||||
if (op === 'append') {
|
||||
return { tokens, nodes: [...nodes, { seq: event.seq, tokens }], deltaTokens: tokens }
|
||||
}
|
||||
const startIdx = nodes.findIndex(node => node.seq === op.start)
|
||||
const endIdx = nodes.findIndex(node => node.seq === op.end)
|
||||
if (startIdx === -1 || endIdx === -1 || startIdx > endIdx) {
|
||||
throw new Error(
|
||||
`token surface: replace at seq ${event.seq} has invalid current range ${op.start}-${op.end}`,
|
||||
)
|
||||
}
|
||||
const removed = nodes
|
||||
.slice(startIdx, endIdx + 1)
|
||||
.reduce((total, node) => total + node.tokens, 0)
|
||||
const next = [...nodes]
|
||||
next.splice(startIdx, endIdx - startIdx + 1, { seq: event.seq, tokens })
|
||||
return { tokens, nodes: next, deltaTokens: tokens - removed }
|
||||
}
|
||||
94
packages/llm/token-meter/src/surface-projection.ts
Normal file
94
packages/llm/token-meter/src/surface-projection.ts
Normal file
@@ -0,0 +1,94 @@
|
||||
/**
|
||||
* The O(1) surface-token fold shared by the token-meter projection units.
|
||||
*
|
||||
* A projection state must stay bounded — the persisted projection cache
|
||||
* checkpoints every unit's whole state, so carrying the priced surface
|
||||
* (one node per model-visible message) would grow a checkpoint without
|
||||
* bound over the session's life. Instead, replacements ride the compact
|
||||
* seam's shadow-price protocol: the metering event immediately before a
|
||||
* surface `replace` (`compact/summary` or `compact/prune`) states the
|
||||
* heuristic price of the exact replaced range, so the fold keeps a running
|
||||
* total plus at most one pending claim and never retains per-node prices.
|
||||
* The counts are exact by construction: producers derive them from the same
|
||||
* fixed estimator this module prices appends with. A replacement without an
|
||||
* armed claim folds with zero delta because bounded state cannot reconstruct
|
||||
* the replaced range; this preserves replay at the cost of possible drift.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-token-meter/surface-projection
|
||||
*/
|
||||
|
||||
import { deriveEventMessage, isSurfaceEvent } from '@deepseek-ai/dsh-session'
|
||||
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
// Type-only: the `compact/*` SessionEventMap merges (shadow-price events).
|
||||
import type {} from '@deepseek-ai/dsh-compact'
|
||||
import { estimateMessage } from './estimate.ts'
|
||||
|
||||
/**
|
||||
* One armed shadow price: the heuristic tokens of the surface range the
|
||||
* IMMEDIATELY following event replaces. Plain JSON — it is part of the
|
||||
* persisted unit state while armed.
|
||||
*/
|
||||
export interface ShadowPriceClaim {
|
||||
/** Declared inclusive first surface-node seq of the priced range. */
|
||||
start: number
|
||||
/** Declared inclusive last surface-node seq of the priced range. */
|
||||
end: number
|
||||
/** Heuristic tokens of the priced range under the fixed estimator. */
|
||||
tokens: number
|
||||
}
|
||||
|
||||
/** One event's effect on a running surface-token total. */
|
||||
export interface SurfaceTokensFold {
|
||||
/** Signed change in the surface total; 0 for events off the surface. */
|
||||
readonly deltaTokens: number
|
||||
/** Claim to carry into the next event; undefined when none survives. */
|
||||
readonly claim: ShadowPriceClaim | undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Fold one committed event onto a running surface-token total.
|
||||
*
|
||||
* A shadow-price event arms a claim; any other event expires it, and a
|
||||
* surface `replace` consumes the claim naming its exact range — the
|
||||
* producers append the metering event and the replacement synchronously
|
||||
* adjacent, so a surviving claim always prices the very next event.
|
||||
* A replace with no claim folds with zero delta because the bounded state
|
||||
* cannot reconstruct the replaced range. An armed claim for another range
|
||||
* still fails because the adjacent events contradict each other.
|
||||
* @param claim - the claim armed by the immediately preceding event, if any.
|
||||
* @param event - the next committed session event.
|
||||
* @returns the signed token delta and the claim state after this event.
|
||||
* @throws when a replacement arrives with an armed claim for a different
|
||||
* range — the metering event was adjacent, so this is a live producer's
|
||||
* shadow-price contract violation, not historical data, and must fail
|
||||
* loud rather than let the total drift.
|
||||
*/
|
||||
export function foldSurfaceProjection(
|
||||
claim: ShadowPriceClaim | undefined,
|
||||
event: SessionEvent,
|
||||
): SurfaceTokensFold {
|
||||
if (event.type === 'compact/summary' || event.type === 'compact/prune') {
|
||||
const { shadowedRange, shadowedTokenCount } = event.data
|
||||
return {
|
||||
deltaTokens: 0,
|
||||
claim: { start: shadowedRange.start, end: shadowedRange.end, tokens: shadowedTokenCount },
|
||||
}
|
||||
}
|
||||
if (!isSurfaceEvent(event)) return { deltaTokens: 0, claim: undefined }
|
||||
const message = deriveEventMessage(event)
|
||||
const tokens = message === null ? 0 : estimateMessage(message)
|
||||
const op = event.surfaceOp
|
||||
if (op === 'append') return { deltaTokens: tokens, claim: undefined }
|
||||
// Sessions recorded before the shadow-price protocol log replacements with
|
||||
// no adjacent metering event; the bounded state cannot reconstruct the
|
||||
// replaced range's price, so fold those neutrally — historical replay
|
||||
// degrades to drift instead of failing.
|
||||
if (claim === undefined) return { deltaTokens: 0, claim: undefined }
|
||||
if (claim.start !== op.start || claim.end !== op.end) {
|
||||
throw new Error(
|
||||
`token surface: replace at seq ${event.seq} over range ${op.start}-${op.end} has no adjacent shadow price`
|
||||
+ ` (armed claim covers ${claim.start}-${claim.end})`,
|
||||
)
|
||||
}
|
||||
return { deltaTokens: tokens - claim.tokens, claim: undefined }
|
||||
}
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
import type { TokenUsage } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
export type { ContextPressureProjection, TokenUsageProjection } from './projection.ts'
|
||||
export type { ContextBreakdownProjection, ContextPressureProjection, TokenUsageProjection } from './projection.ts'
|
||||
|
||||
/** Token-meter plugin configuration; the fixed estimator has no settings. */
|
||||
export type TokenMeterConfig = Record<string, never>
|
||||
|
||||
@@ -4,8 +4,11 @@
|
||||
|
||||
import { z } from 'zod'
|
||||
import type { TokenUsage } from '@deepseek-ai/dsh-llm'
|
||||
import type { SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection'
|
||||
import type { ContextPressureProjection, TokenUsageProjection } from './projection.ts'
|
||||
import { foldSurfaceProjection } from './surface-projection.ts'
|
||||
import type { ShadowPriceClaim } from './surface-projection.ts'
|
||||
|
||||
interface UsageSample {
|
||||
turn: number
|
||||
@@ -60,6 +63,7 @@ const projectionSchema = z.object({
|
||||
// `number | undefined` where the interface declares absent-or-number fields.
|
||||
const pressureSchema = z.object({
|
||||
pressureTokens: z.number().int().nonnegative().optional(),
|
||||
projectedTokens: z.number().int().nonnegative().optional(),
|
||||
contextWindow: z.number().int().positive().optional(),
|
||||
}).strict() as unknown as z.ZodType<ContextPressureProjection>
|
||||
|
||||
@@ -67,6 +71,29 @@ const pressureSchema = z.object({
|
||||
const pressureFrom = (usage: TokenUsage): number =>
|
||||
usage.inputTokens + (usage.cacheReadTokens ?? 0) + (usage.cacheWriteTokens ?? 0)
|
||||
|
||||
/** The usage a chunk or finalized message reports for its step, if any. */
|
||||
const usageOf = (event: SessionEvent): TokenUsage | undefined =>
|
||||
event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
|
||||
? event.data.chunk.usage
|
||||
: event.type === 'assistant/message'
|
||||
? event.data.usage
|
||||
: undefined
|
||||
|
||||
/**
|
||||
* Context-occupancy state: the two independent last-wins records plus the
|
||||
* O(1) running surface total needed to carry the newest sample forward.
|
||||
*/
|
||||
interface ContextPressureState {
|
||||
contextWindow?: number
|
||||
pressureTokens?: number
|
||||
/** Running heuristic total over the current surface ({@link foldSurfaceProjection}). */
|
||||
surfaceTokens: number
|
||||
/** {@link surfaceTokens} at the newest usage sample; absent until one lands. */
|
||||
sampledSurfaceTokens?: number
|
||||
/** Shadow price armed by the immediately preceding metering event. */
|
||||
claim?: ShadowPriceClaim
|
||||
}
|
||||
|
||||
/**
|
||||
* Token-meter's session projection unit.
|
||||
*
|
||||
@@ -115,39 +142,65 @@ ProjectionDefinition<'tokenUsage', TokenUsageState> = {
|
||||
/**
|
||||
* Token-meter's context-occupancy projection unit.
|
||||
*
|
||||
* Two independent last-wins slots: the newest usage sample supplies the
|
||||
* Independent last-wins slots: the newest usage sample supplies the provider
|
||||
* numerator, the newest `request/context` record the denominator. Both are
|
||||
* whole values, so replay order alone decides the result and no cross-field
|
||||
* consistency is claimed — the pair is explicitly not one atomic request
|
||||
* observation (see {@link ContextPressureProjection}).
|
||||
*
|
||||
* The numerator is prompt-side only, so it holds still while a turn streams
|
||||
* and steps forward once the next request reports its usage.
|
||||
* `pressureTokens` is prompt-side only, so it holds still while a turn streams
|
||||
* and steps forward once the next request reports its usage. Because nothing
|
||||
* but a request reports usage, it also cannot see a compaction: the fold
|
||||
* therefore carries a running surface total alongside it and publishes
|
||||
* `projectedTokens` — the sample plus the surface's signed movement since it
|
||||
* was taken — so occupancy answers for the next request rather than the last
|
||||
* one. The total rides {@link foldSurfaceProjection}, so the state stays O(1)
|
||||
* and a replacement shrinks it by its logged shadow price. A replacement
|
||||
* without a claim preserves the previous total. A usage sample is stamped
|
||||
* BEFORE the same event joins the surface, so an `assistant/message` anchors
|
||||
* against the surface its own request saw.
|
||||
*/
|
||||
export const contextPressureProjectionDefinition:
|
||||
ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
|
||||
ProjectionDefinition<'contextPressure', ContextPressureState> = {
|
||||
key: 'contextPressure',
|
||||
schema: pressureSchema,
|
||||
init: () => ({}),
|
||||
init: () => ({ surfaceTokens: 0 }),
|
||||
apply: (state, event) => {
|
||||
const fold = foldSurfaceProjection(state.claim, event)
|
||||
let next = state
|
||||
if (event.type === 'request/context') {
|
||||
const contextWindow = event.data.contextWindow
|
||||
if (contextWindow === state.contextWindow) return state
|
||||
if (contextWindow !== undefined) return { ...state, contextWindow }
|
||||
const { contextWindow: _removed, ...withoutContextWindow } = state
|
||||
return withoutContextWindow
|
||||
if (contextWindow !== state.contextWindow) {
|
||||
if (contextWindow !== undefined) {
|
||||
next = { ...next, contextWindow }
|
||||
} else {
|
||||
const { contextWindow: _removed, ...withoutContextWindow } = next
|
||||
next = withoutContextWindow
|
||||
}
|
||||
}
|
||||
}
|
||||
const usage = event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
|
||||
? event.data.chunk.usage
|
||||
: event.type === 'assistant/message'
|
||||
? event.data.usage
|
||||
: undefined
|
||||
if (usage === undefined) return state
|
||||
const pressureTokens = pressureFrom(usage)
|
||||
return pressureTokens === state.pressureTokens
|
||||
? state
|
||||
: { ...state, pressureTokens }
|
||||
const usage = usageOf(event)
|
||||
if (usage !== undefined) {
|
||||
const pressureTokens = pressureFrom(usage)
|
||||
if (pressureTokens !== next.pressureTokens || next.sampledSurfaceTokens !== next.surfaceTokens) {
|
||||
next = { ...next, pressureTokens, sampledSurfaceTokens: next.surfaceTokens }
|
||||
}
|
||||
}
|
||||
if (fold.deltaTokens !== 0) {
|
||||
next = { ...next, surfaceTokens: next.surfaceTokens + fold.deltaTokens }
|
||||
}
|
||||
// A defined fold.claim is always freshly built, so presence decides claim
|
||||
// bookkeeping: no claim before or after this event leaves `next` as is.
|
||||
if (state.claim === undefined && fold.claim === undefined) return next
|
||||
const { claim: _expired, ...withoutClaim } = next
|
||||
return fold.claim === undefined ? withoutClaim : { ...withoutClaim, claim: fold.claim }
|
||||
},
|
||||
view: state => state,
|
||||
stateVersion: 2,
|
||||
view: ({ contextWindow, pressureTokens, surfaceTokens, sampledSurfaceTokens }) => ({
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
...pressureTokens === undefined ? {} : { pressureTokens },
|
||||
...pressureTokens === undefined || sampledSurfaceTokens === undefined
|
||||
? {}
|
||||
: { projectedTokens: Math.max(0, pressureTokens + surfaceTokens - sampledSurfaceTokens) },
|
||||
}),
|
||||
stateVersion: 4,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,309 @@
|
||||
// contextBreakdown projection: heuristic system/tools/message composition,
|
||||
// plus the shared estimator's pricing branches.
|
||||
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import { createMessage, createUserMessage } from '@deepseek-ai/dsh-llm'
|
||||
import type { ContentBlock, ToolSchema } from '@deepseek-ai/dsh-llm'
|
||||
import SessionStore from '@deepseek-ai/dsh-session'
|
||||
import type { Session, SessionEvent } from '@deepseek-ai/dsh-session'
|
||||
import SessionProjectionRegistry from '@deepseek-ai/dsh-session-projection'
|
||||
import TokenMeterService from '@deepseek-ai/dsh-token-meter'
|
||||
import type { ContextBreakdownProjection } from '@deepseek-ai/dsh-token-meter/client'
|
||||
import { contextBreakdownProjectionDefinition } from '../src/breakdown-projection.ts'
|
||||
import {
|
||||
estimateContent,
|
||||
estimateHeader,
|
||||
estimateMessage,
|
||||
estimateSystemTokens,
|
||||
estimateToolsTokens,
|
||||
} from '../src/estimate.ts'
|
||||
|
||||
const CONFIG = { provider: 'test', model: 'test-model' }
|
||||
|
||||
const TOOLS: ToolSchema[] = [{
|
||||
name: 'bash',
|
||||
description: 'run a command',
|
||||
parameters: { type: 'object', properties: {} },
|
||||
}]
|
||||
|
||||
async function harness(): Promise<{ ctx: Context; session: Session }> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(SessionProjectionRegistry)
|
||||
await ctx.plugin(TokenMeterService)
|
||||
return { ctx, session: ctx.sessions.create() }
|
||||
}
|
||||
|
||||
const projected = (ctx: Context, session: Session): ContextBreakdownProjection => {
|
||||
const value = ctx.sessionProjections.snapshot(session).values.contextBreakdown
|
||||
if (value === undefined) throw new Error('contextBreakdown projection is not registered')
|
||||
return value
|
||||
}
|
||||
|
||||
function appendUser(session: Session, text: string): number {
|
||||
return session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text }],
|
||||
source: { kind: 'user' },
|
||||
}), { surfaceOp: 'append' }).seq
|
||||
}
|
||||
|
||||
/**
|
||||
* Meter one upcoming replacement the way compact-basic does: price the
|
||||
* replaced span from the measurement service's own nodes and log the
|
||||
* shadow-price event directly before the replace.
|
||||
*/
|
||||
function appendSummaryMeter(ctx: Context, session: Session, start: number, end: number): void {
|
||||
const nodes = ctx.tokenMeter.measure(session).nodes
|
||||
const startIdx = nodes.findIndex(node => node.seq === start)
|
||||
const endIdx = nodes.findIndex(node => node.seq === end)
|
||||
const shadowed = nodes.slice(startIdx, endIdx + 1)
|
||||
session.append('compact/summary', {
|
||||
summary: [{ type: 'text', text: 'summary' }],
|
||||
shadowedRange: { start, end },
|
||||
shadowedSeqs: shadowed.map(node => node.seq),
|
||||
shadowedTokenCount: shadowed.reduce((total, node) => total + node.tokens, 0),
|
||||
provider: 'mock',
|
||||
model: 'mock',
|
||||
})
|
||||
}
|
||||
|
||||
describe('contextBreakdown session projection', () => {
|
||||
it('serves zeros for an empty log', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
expect(projected(ctx, session)).toEqual({ systemTokens: 0, toolsTokens: 0, messageTokens: 0 })
|
||||
})
|
||||
|
||||
it('prices the newest envelope last-wins and pushes no change for a restated one', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
session.append('request/header', {
|
||||
header: { config: CONFIG, system: 'You are terse.', tools: TOOLS },
|
||||
reason: 'initial',
|
||||
})
|
||||
expect(projected(ctx, session)).toEqual({
|
||||
systemTokens: estimateSystemTokens({ config: CONFIG, system: 'You are terse.' }),
|
||||
toolsTokens: estimateToolsTokens({ config: CONFIG, tools: TOOLS }),
|
||||
messageTokens: 0,
|
||||
})
|
||||
|
||||
const changed: string[] = []
|
||||
ctx.sessionProjections.onChanged((_session, key) => { changed.push(key) })
|
||||
session.append('request/header', {
|
||||
header: { config: CONFIG, system: 'You are terse.', tools: TOOLS },
|
||||
reason: 'change',
|
||||
})
|
||||
session.append('todo/write', { todos: [] })
|
||||
expect(changed).not.toContain('contextBreakdown')
|
||||
|
||||
// A system-less, tool-less envelope prices back to zero.
|
||||
session.append('request/header', { header: { config: CONFIG }, reason: 'change' })
|
||||
expect(projected(ctx, session)).toEqual({ systemTokens: 0, toolsTokens: 0, messageTokens: 0 })
|
||||
})
|
||||
|
||||
it('sums surface appends and skips an empty-content assistant message', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
appendUser(session, 'abcd')
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
session.append('assistant/message', {
|
||||
turn: 1,
|
||||
step: 1,
|
||||
message: createMessage({
|
||||
role: 'assistant',
|
||||
content: [],
|
||||
source: { kind: 'model', provider: 'mock', model: 'mock' },
|
||||
}),
|
||||
usage: { inputTokens: 9, outputTokens: 0 },
|
||||
}, { surfaceOp: 'append', sourceEventSeqs: [] })
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
// 'abcd' prices to 9 (1 text + 4 block + 4 role); the usage-only assistant
|
||||
// message derives to no transcript entry and adds nothing.
|
||||
expect(projected(ctx, session).messageTokens).toBe(9)
|
||||
})
|
||||
|
||||
it('shrinks the message figure when a metered replacement compacts the surface', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
const first = appendUser(session, 'before compaction, a longer message')
|
||||
const second = appendUser(session, 'and a second entry')
|
||||
const summary = createUserMessage({
|
||||
content: [{ type: 'text', text: 'summary' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
})
|
||||
appendSummaryMeter(ctx, session, first, second)
|
||||
session.append('user/message', summary, {
|
||||
surfaceOp: { op: 'replace', start: first, end: second },
|
||||
sourceEventSeqs: [first, second],
|
||||
})
|
||||
expect(projected(ctx, session).messageTokens).toBe(estimateMessage(summary))
|
||||
})
|
||||
|
||||
it('keeps the message figure equal to the service surface across appends and a compaction', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
// The panel's composition rows and `measure()` answer the same question in
|
||||
// the same vocabulary; one shared fold is what makes that true.
|
||||
const agree = (): number => {
|
||||
const messageTokens = projected(ctx, session).messageTokens
|
||||
expect(messageTokens).toBe(ctx.tokenMeter.measure(session).surfaceTokens)
|
||||
return messageTokens
|
||||
}
|
||||
session.append('request/header', {
|
||||
header: { config: CONFIG, system: 'You are terse.', tools: TOOLS },
|
||||
reason: 'initial',
|
||||
})
|
||||
expect(agree()).toBe(0)
|
||||
|
||||
const question = appendUser(session, 'a first question, long enough to price above zero')
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
const answer = session.append('assistant/message', {
|
||||
turn: 1,
|
||||
step: 1,
|
||||
message: createMessage({
|
||||
role: 'assistant',
|
||||
content: [{ type: 'text', text: 'a considered answer' }],
|
||||
source: { kind: 'model', provider: 'mock', model: 'mock' },
|
||||
}),
|
||||
usage: { inputTokens: 40, outputTokens: 7 },
|
||||
}, { surfaceOp: 'append', sourceEventSeqs: [] }).seq
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
const grown = agree()
|
||||
expect(grown).toBeGreaterThan(0)
|
||||
|
||||
appendSummaryMeter(ctx, session, question, answer)
|
||||
// The armed shadow price must not move the published figure by itself.
|
||||
expect(agree()).toBe(grown)
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'summary' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
}), {
|
||||
surfaceOp: { op: 'replace', start: question, end: answer },
|
||||
sourceEventSeqs: [question, answer],
|
||||
})
|
||||
expect(agree()).toBeLessThan(grown)
|
||||
})
|
||||
|
||||
it('folds a replacement without a claim at zero and fails on a mismatched claim', () => {
|
||||
const definition = contextBreakdownProjectionDefinition
|
||||
const replace = (start: number, end: number): SessionEvent => ({
|
||||
type: 'user/message',
|
||||
seq: 9,
|
||||
time: 0,
|
||||
data: createUserMessage({ content: [{ type: 'text', text: 'x' }], source: { kind: 'user' } }),
|
||||
surfaceOp: { op: 'replace', start, end },
|
||||
sourceEventSeqs: [start, end],
|
||||
} as unknown as SessionEvent)
|
||||
const append = (seq: number): SessionEvent => ({
|
||||
type: 'user/message',
|
||||
seq,
|
||||
time: 0,
|
||||
data: createUserMessage({ content: [{ type: 'text', text: 'x' }], source: { kind: 'user' } }),
|
||||
surfaceOp: 'append',
|
||||
} as unknown as SessionEvent)
|
||||
const meter = (start: number, end: number, seq: number): SessionEvent => ({
|
||||
type: 'compact/prune',
|
||||
seq,
|
||||
time: 0,
|
||||
data: { shadowedRange: { start, end }, shadowedSeqs: [start, end], shadowedTokenCount: 5 },
|
||||
} as unknown as SessionEvent)
|
||||
let state = definition.init()
|
||||
state = definition.apply(state, append(1))
|
||||
state = definition.apply(state, append(3))
|
||||
// No metering event: the replacement contributes zero instead of throwing.
|
||||
expect(definition.view(definition.apply(state, replace(1, 3))).messageTokens)
|
||||
.toBe(definition.view(state).messageTokens)
|
||||
// An adjacent claim for another range contradicts the replacement.
|
||||
const mismatched = definition.apply(state, meter(1, 1, 8))
|
||||
expect(() => definition.apply(mismatched, replace(1, 3))).toThrow('no adjacent shadow price')
|
||||
// A claim expires after one intervening event, so replacement delta is zero.
|
||||
let expired = definition.apply(state, meter(1, 3, 8))
|
||||
expired = definition.apply(expired, { type: 'todo/write', seq: 9, time: 0, data: { todos: [] } } as unknown as SessionEvent)
|
||||
expect(definition.view(definition.apply(expired, replace(1, 3))).messageTokens)
|
||||
.toBe(definition.view(state).messageTokens)
|
||||
// The armed claim prices exactly the next event's matching replacement.
|
||||
const armed = definition.apply(state, meter(1, 3, 8))
|
||||
expect(definition.view(definition.apply(armed, replace(1, 3))).messageTokens)
|
||||
.toBe(definition.view(state).messageTokens - 5 + estimateMessage(
|
||||
createUserMessage({ content: [{ type: 'text', text: 'x' }], source: { kind: 'user' } }),
|
||||
))
|
||||
})
|
||||
|
||||
it('keeps the persisted checkpoint O(1) as the surface grows and compacts', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
const first = appendUser(session, 'the first of many messages')
|
||||
for (let index = 0; index < 24; index += 1) appendUser(session, `message number ${index} with some text`)
|
||||
const last = appendUser(session, 'the last message before compaction')
|
||||
const stateKeys = (): string[] => {
|
||||
const row = ctx.sessionProjections.checkpoint(session)['contextBreakdown']
|
||||
if (row === undefined) throw new Error('contextBreakdown checkpoint row is missing')
|
||||
return Object.keys(row.val as Record<string, unknown>).sort()
|
||||
}
|
||||
// Growth adds no per-node bookkeeping to the durable state.
|
||||
expect(stateKeys()).toEqual(['messageTokens', 'systemTokens', 'toolsTokens'])
|
||||
const shadowed = session.surface.nodes.slice(
|
||||
session.surface.nodes.indexOf(first),
|
||||
session.surface.nodes.indexOf(last) + 1,
|
||||
)
|
||||
appendSummaryMeter(ctx, session, first, last)
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'summary' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
}), {
|
||||
surfaceOp: { op: 'replace', start: first, end: last },
|
||||
sourceEventSeqs: [...shadowed],
|
||||
})
|
||||
expect(stateKeys()).toEqual(['messageTokens', 'systemTokens', 'toolsTokens'])
|
||||
expect(projected(ctx, session).messageTokens)
|
||||
.toBe(ctx.tokenMeter.measure(session).surfaceTokens)
|
||||
})
|
||||
|
||||
it('restores from a JSON checkpoint and unregisters with the token-meter fiber', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SessionStore)
|
||||
await ctx.plugin(SessionProjectionRegistry)
|
||||
const meterFiber = await ctx.plugin(TokenMeterService)
|
||||
const session = ctx.sessions.create()
|
||||
session.append('request/header', {
|
||||
header: { config: CONFIG, system: 'You are terse.' },
|
||||
reason: 'initial',
|
||||
})
|
||||
appendUser(session, 'abcd')
|
||||
const checkpoint = JSON.parse(JSON.stringify(
|
||||
ctx.sessionProjections.checkpoint(session),
|
||||
)) as ReturnType<typeof ctx.sessionProjections.checkpoint>
|
||||
|
||||
await meterFiber.dispose()
|
||||
expect(ctx.sessionProjections.snapshot(session).values).not.toHaveProperty('contextBreakdown')
|
||||
|
||||
await ctx.plugin(TokenMeterService)
|
||||
expect(ctx.sessionProjections.viewCheckpoint(checkpoint).contextBreakdown).toEqual({
|
||||
systemTokens: estimateSystemTokens({ config: CONFIG, system: 'You are terse.' }),
|
||||
toolsTokens: 0,
|
||||
messageTokens: 9,
|
||||
})
|
||||
})
|
||||
})
|
||||
|
||||
describe('shared estimator', () => {
|
||||
it('prices every content-block shape under the fixed heuristic', () => {
|
||||
expect(estimateContent([{ type: 'text', text: 'abcd' }])).toBe(5)
|
||||
expect(estimateContent([{ type: 'reasoning', text: 'abcdefgh' }] as ContentBlock[])).toBe(6)
|
||||
expect(estimateContent([{ type: 'tool-call', id: 'c' as never, name: 'bash', arguments: '{"a":1}' }])).toBe(7)
|
||||
expect(estimateContent([{
|
||||
type: 'tool-result', toolCallId: 'c' as never,
|
||||
content: [{ type: 'text', text: 'abcd' }],
|
||||
}])).toBe(9)
|
||||
const unknown = { type: 'mystery', payload: 'abc' } as unknown as ContentBlock
|
||||
expect(estimateContent([unknown])).toBe(4 + Math.ceil(JSON.stringify(unknown).length / 4))
|
||||
})
|
||||
|
||||
it('prices envelope parts independently and absent parts to zero', () => {
|
||||
expect(estimateSystemTokens(undefined)).toBe(0)
|
||||
expect(estimateSystemTokens({ config: CONFIG })).toBe(0)
|
||||
expect(estimateSystemTokens({ config: CONFIG, system: 'abcdefgh' })).toBe(6)
|
||||
expect(estimateToolsTokens(undefined)).toBe(0)
|
||||
expect(estimateToolsTokens({ config: CONFIG, tools: [] })).toBe(0)
|
||||
expect(estimateToolsTokens({ config: CONFIG, tools: TOOLS }))
|
||||
.toBe(Math.ceil(JSON.stringify(TOOLS).length / 4) + 4)
|
||||
expect(estimateHeader(undefined)).toBe(0)
|
||||
expect(estimateHeader({ config: CONFIG, system: 'abcdefgh', tools: TOOLS }))
|
||||
.toBe(6 + Math.ceil(JSON.stringify(TOOLS).length / 4) + 4)
|
||||
})
|
||||
})
|
||||
@@ -147,7 +147,7 @@ describe('TokenMeterService pricing', () => {
|
||||
|
||||
it('returns a detached deeply immutable empty measurement', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('empty'))
|
||||
const session = Session.create(SessionId('empty'))
|
||||
const result = service.measure(session)
|
||||
expect(result).toEqual({
|
||||
logRevision: 0,
|
||||
@@ -168,7 +168,7 @@ describe('TokenMeterService pricing', () => {
|
||||
|
||||
it('keeps an earlier unified snapshot detached from later replay', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('detached'))
|
||||
const session = Session.create(SessionId('detached'))
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'first' }],
|
||||
source: { kind: 'user' },
|
||||
@@ -200,7 +200,7 @@ describe('TokenMeterService pricing', () => {
|
||||
|
||||
it('prices header, tools, and surface when no reusable usage exists', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('heuristic'))
|
||||
const session = Session.create(SessionId('heuristic'))
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'question' }],
|
||||
source: { kind: 'user' },
|
||||
@@ -218,7 +218,7 @@ describe('TokenMeterService pricing', () => {
|
||||
|
||||
it('keeps request-header overrides out of the returned surface', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('override-surface'))
|
||||
const session = Session.create(SessionId('override-surface'))
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'question' }],
|
||||
source: { kind: 'user' },
|
||||
@@ -246,7 +246,7 @@ describe('replay anchors and surface folds', () => {
|
||||
|
||||
it('uses disjoint provider usage and signed durable-output rewrites', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('usage'))
|
||||
const session = Session.create(SessionId('usage'))
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'before' }],
|
||||
source: { kind: 'user' },
|
||||
@@ -267,7 +267,7 @@ describe('replay anchors and surface folds', () => {
|
||||
|
||||
it('selects a heuristic anchor when provider usage would undercut its scale', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('low-usage-anchor'))
|
||||
const session = Session.create(SessionId('low-usage-anchor'))
|
||||
const system = 'system context'
|
||||
const requestHeader = header('deepseek-v4-flash', { system })
|
||||
appendSuccessfulCall(session, requestHeader, {
|
||||
@@ -297,7 +297,7 @@ describe('replay anchors and surface folds', () => {
|
||||
|
||||
it('uses an estimated anchor when provider usage is absent', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('missing-usage'))
|
||||
const session = Session.create(SessionId('missing-usage'))
|
||||
appendSuccessfulCall(session, header('deepseek-v4-flash', { system: 's' }), {
|
||||
providerText: 'provider',
|
||||
durableText: 'rewritten',
|
||||
@@ -314,8 +314,8 @@ describe('replay anchors and surface folds', () => {
|
||||
})
|
||||
|
||||
it('distinguishes explicit empty provenance from absent legacy provenance', () => {
|
||||
const explicit = new Session(SessionId('explicit-empty'))
|
||||
const legacy = new Session(SessionId('legacy-absent'))
|
||||
const explicit = Session.create(SessionId('explicit-empty'))
|
||||
const legacy = Session.create(SessionId('legacy-absent'))
|
||||
appendSuccessfulCall(explicit, header('deepseek-v4-flash'), {
|
||||
durableText: 'listener injected text',
|
||||
providerText: '',
|
||||
@@ -335,7 +335,7 @@ describe('replay anchors and surface folds', () => {
|
||||
|
||||
it('keeps only the latest successful request anchor across model switches', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('switch'))
|
||||
const session = Session.create(SessionId('switch'))
|
||||
const alphaHeader = header('alpha', { system: 'same envelope' })
|
||||
appendSuccessfulCall(session, alphaHeader, { usage: USAGE, providerText: 'alpha' })
|
||||
expect(service.measure(session).baseline).toMatchObject({ kind: 'usage', tokens: 34 })
|
||||
@@ -356,7 +356,7 @@ describe('replay anchors and surface folds', () => {
|
||||
|
||||
it('invalidates usage for any canonical envelope change or explicit override', () => {
|
||||
const service = meter()
|
||||
const session = new Session(SessionId('envelope'))
|
||||
const session = Session.create(SessionId('envelope'))
|
||||
const anchoredHeader = header('deepseek-v4-flash', { system: 'one' })
|
||||
appendSuccessfulCall(session, anchoredHeader, { usage: USAGE })
|
||||
expect(service.measure(session, { ...anchoredHeader, tools: [] }).baseline.kind).toBe('usage')
|
||||
@@ -375,7 +375,7 @@ describe('replay anchors and surface folds', () => {
|
||||
})
|
||||
|
||||
it('folds the latest full header snapshot into the effective envelope', () => {
|
||||
const session = new Session(SessionId('header-snapshot'))
|
||||
const session = Session.create(SessionId('header-snapshot'))
|
||||
appendHeader(session, header('deepseek-v4-flash'))
|
||||
session.append('request/header', {
|
||||
header: header('deepseek-v4-pro'),
|
||||
@@ -388,7 +388,7 @@ describe('replay anchors and surface folds', () => {
|
||||
|
||||
it('replays seeded append and replace operations with signed deltas', () => {
|
||||
const service = meter()
|
||||
const original = new Session(SessionId('surface-original'))
|
||||
const original = Session.create(SessionId('surface-original'))
|
||||
appendSuccessfulCall(original, header('deepseek-v4-flash'), {
|
||||
usage: USAGE,
|
||||
providerText: 'long provider answer '.repeat(100),
|
||||
@@ -397,7 +397,7 @@ describe('replay anchors and surface folds', () => {
|
||||
content: [{ type: 'text', text: 'new tail' }],
|
||||
source: { kind: 'user' },
|
||||
}), { surfaceOp: 'append' })
|
||||
const seeded = new Session(SessionId('surface-seeded'), original.events)
|
||||
const seeded = Session.create(SessionId('surface-seeded'), original.events)
|
||||
const before = service.measure(seeded)
|
||||
expect(before.nodes).toHaveLength(2)
|
||||
expect(before.surfaceDeltaTokens).toBeGreaterThan(0)
|
||||
@@ -423,7 +423,7 @@ describe('replay anchors and surface folds', () => {
|
||||
})
|
||||
|
||||
it('prices an empty assistant surface anchor as zero', () => {
|
||||
const session = new Session(SessionId('empty-assistant'))
|
||||
const session = Session.create(SessionId('empty-assistant'))
|
||||
appendSuccessfulCall(session, header('deepseek-v4-flash'), {
|
||||
providerText: '',
|
||||
durableText: '',
|
||||
@@ -444,7 +444,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
}
|
||||
|
||||
it('rejects an assistant without its step boundary transactionally', () => {
|
||||
const session = new Session(SessionId('bad-step'))
|
||||
const session = Session.create(SessionId('bad-step'))
|
||||
appendHeader(session, header('deepseek-v4-flash'))
|
||||
session.append('assistant/message', {
|
||||
turn: 1,
|
||||
@@ -462,7 +462,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
})
|
||||
|
||||
it('clears completed step boundaries and rejects overlapping or late step events', () => {
|
||||
const overlapping = new Session(SessionId('overlapping-step'))
|
||||
const overlapping = Session.create(SessionId('overlapping-step'))
|
||||
overlapping.append('step/start', { turn: 1, step: 1 })
|
||||
overlapping.append('step/start', { turn: 1, step: 2 })
|
||||
expectRepeatedFailure(
|
||||
@@ -471,7 +471,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
/arrived before turn 1\/step 1 ended/,
|
||||
)
|
||||
|
||||
const late = new Session(SessionId('late-assistant'))
|
||||
const late = Session.create(SessionId('late-assistant'))
|
||||
late.append('step/start', { turn: 1, step: 1 })
|
||||
appendHeader(late, header('deepseek-v4-flash'))
|
||||
late.append('step/end', { turn: 1, step: 1 })
|
||||
@@ -493,7 +493,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
/no matching step\/start/,
|
||||
)
|
||||
|
||||
const mismatchedEnd = new Session(SessionId('mismatched-end'))
|
||||
const mismatchedEnd = Session.create(SessionId('mismatched-end'))
|
||||
mismatchedEnd.append('step/start', { turn: 1, step: 1 })
|
||||
mismatchedEnd.append('step/end', { turn: 1, step: 2 })
|
||||
expectRepeatedFailure(
|
||||
@@ -532,7 +532,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
},
|
||||
]
|
||||
for (const testCase of cases) {
|
||||
const session = new Session(SessionId(`bad-source-${testCase.name}`))
|
||||
const session = Session.create(SessionId(`bad-source-${testCase.name}`))
|
||||
session.append('step/start', { turn: 1, step: 1 })
|
||||
appendHeader(session, header('deepseek-v4-flash'))
|
||||
const sourceEventSeqs = testCase.appendSource(session)
|
||||
@@ -554,7 +554,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
})
|
||||
|
||||
it('rejects repeated and non-earlier assistant provenance', () => {
|
||||
const duplicate = new Session(SessionId('duplicate-source'))
|
||||
const duplicate = Session.create(SessionId('duplicate-source'))
|
||||
duplicate.append('step/start', { turn: 1, step: 1 })
|
||||
appendHeader(duplicate, header('deepseek-v4-flash'))
|
||||
const source = duplicate.append('assistant/chunk', {
|
||||
@@ -584,7 +584,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
})
|
||||
expect(() => meter().measure(duplicate)).toThrow(/repeats source seq/)
|
||||
|
||||
const future = new Session(SessionId('future-source'))
|
||||
const future = Session.create(SessionId('future-source'))
|
||||
future.append('step/start', { turn: 1, step: 1 })
|
||||
appendHeader(future, header('deepseek-v4-flash'))
|
||||
appendUnchecked(future, {
|
||||
@@ -611,7 +611,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
})
|
||||
|
||||
it('does not partially apply a malformed assistant replacement', () => {
|
||||
const session = new Session(SessionId('transactional-replace'))
|
||||
const session = Session.create(SessionId('transactional-replace'))
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'head' }],
|
||||
source: { kind: 'user' },
|
||||
@@ -638,7 +638,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
})
|
||||
|
||||
it('rejects corrupt replacement ranges without advancing the replay cursor', () => {
|
||||
const session = new Session(SessionId('bad-replace'))
|
||||
const session = Session.create(SessionId('bad-replace'))
|
||||
const head = session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'head' }],
|
||||
source: { kind: 'user' },
|
||||
@@ -671,7 +671,7 @@ describe('malformed replay and listener lifecycle', () => {
|
||||
type: 'turn/start',
|
||||
seq: 0,
|
||||
time: 1,
|
||||
data: { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } },
|
||||
data: { turn: 1 },
|
||||
}] })
|
||||
activeMeter.measure(session)
|
||||
session.append('user/message', createUserMessage({
|
||||
|
||||
@@ -70,6 +70,26 @@ const projected = (ctx: Context, session: Session): TokenUsageProjection => {
|
||||
return value
|
||||
}
|
||||
|
||||
/**
|
||||
* Meter one upcoming replacement the way compact-basic does: price the
|
||||
* replaced span from the measurement service's own nodes and log the
|
||||
* shadow-price event directly before the replace.
|
||||
*/
|
||||
function appendSummaryMeter(ctx: Context, session: Session, start: number, end: number): void {
|
||||
const nodes = ctx.tokenMeter.measure(session).nodes
|
||||
const startIdx = nodes.findIndex(node => node.seq === start)
|
||||
const endIdx = nodes.findIndex(node => node.seq === end)
|
||||
const shadowed = nodes.slice(startIdx, endIdx + 1)
|
||||
session.append('compact/summary', {
|
||||
summary: [{ type: 'text', text: 'summary' }],
|
||||
shadowedRange: { start, end },
|
||||
shadowedSeqs: shadowed.map(node => node.seq),
|
||||
shadowedTokenCount: shadowed.reduce((total, node) => total + node.tokens, 0),
|
||||
provider: 'mock',
|
||||
model: 'mock',
|
||||
})
|
||||
}
|
||||
|
||||
describe('tokenUsage session projection', () => {
|
||||
it('serves zero buckets for an empty log', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
@@ -184,6 +204,7 @@ describe('tokenUsage session projection', () => {
|
||||
content: [{ type: 'text', text: 'before compaction' }],
|
||||
source: { kind: 'user' },
|
||||
}), { surfaceOp: 'append' })
|
||||
appendSummaryMeter(ctx, session, before.seq, before.seq)
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'compacted' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
@@ -235,6 +256,34 @@ function recordContext(session: Session, model: string, contextWindow?: number):
|
||||
})
|
||||
}
|
||||
|
||||
/** Append one model-visible user turn and return its surface seq. */
|
||||
function appendUser(session: Session, text: string): number {
|
||||
return session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text }],
|
||||
source: { kind: 'user' },
|
||||
}), { surfaceOp: 'append' }).seq
|
||||
}
|
||||
|
||||
/** Append one finalized assistant turn carrying its provider usage. */
|
||||
function appendAssistant(
|
||||
session: Session,
|
||||
text: string,
|
||||
usage: TokenUsage,
|
||||
turn: number,
|
||||
step: number,
|
||||
): number {
|
||||
return session.append('assistant/message', {
|
||||
turn,
|
||||
step,
|
||||
message: createMessage({
|
||||
role: 'assistant',
|
||||
content: [{ type: 'text', text }],
|
||||
source: { kind: 'model', provider: 'mock', model: 'mock' },
|
||||
}),
|
||||
usage,
|
||||
}, { surfaceOp: 'append', sourceEventSeqs: [] }).seq
|
||||
}
|
||||
|
||||
describe('contextPressure session projection', () => {
|
||||
it('serves no pressure or capacity for an empty log', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
@@ -277,9 +326,13 @@ describe('contextPressure session projection', () => {
|
||||
startStep(session, 1, 1)
|
||||
recordContext(session, 'small', 64_000)
|
||||
usageChunk(session, { inputTokens: 100, outputTokens: 10 }, 1, 1)
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100, contextWindow: 64_000 })
|
||||
expect(pressure(ctx, session)).toEqual({
|
||||
pressureTokens: 100, projectedTokens: 100, contextWindow: 64_000,
|
||||
})
|
||||
recordContext(session, 'large', 256_000)
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100, contextWindow: 256_000 })
|
||||
expect(pressure(ctx, session)).toEqual({
|
||||
pressureTokens: 100, projectedTokens: 100, contextWindow: 256_000,
|
||||
})
|
||||
})
|
||||
|
||||
it('removes an older capacity when the newest route advertises none', async () => {
|
||||
@@ -288,7 +341,7 @@ describe('contextPressure session projection', () => {
|
||||
recordContext(session, 'small', 64_000)
|
||||
usageChunk(session, { inputTokens: 100, outputTokens: 10 }, 1, 1)
|
||||
recordContext(session, 'unknown')
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100 })
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100, projectedTokens: 100 })
|
||||
})
|
||||
|
||||
it('pushes no change for unrelated events or a restated capacity', async () => {
|
||||
@@ -319,7 +372,7 @@ describe('contextPressure session projection', () => {
|
||||
const checkpoint = JSON.parse(JSON.stringify(
|
||||
ctx.sessionProjections.checkpoint(session),
|
||||
)) as ReturnType<typeof ctx.sessionProjections.checkpoint>
|
||||
expect(checkpoint.contextPressure?.ver).toBe(2)
|
||||
expect(checkpoint.contextPressure?.ver).toBe(4)
|
||||
|
||||
await meterFiber.dispose()
|
||||
expect(ctx.sessionProjections.snapshot(session).values).not.toHaveProperty('contextPressure')
|
||||
@@ -327,7 +380,81 @@ describe('contextPressure session projection', () => {
|
||||
await ctx.plugin(TokenMeterService)
|
||||
expect(ctx.sessionProjections.viewCheckpoint(checkpoint).contextPressure).toEqual({
|
||||
pressureTokens: 42,
|
||||
projectedTokens: 42,
|
||||
contextWindow: 64_000,
|
||||
})
|
||||
})
|
||||
|
||||
it('carries the sample forward over surface growth and a compaction', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
recordContext(session, 'large', 128_000)
|
||||
const question = appendUser(session, 'a first question worth a few tokens')
|
||||
startStep(session, 1, 1)
|
||||
// The provider prices the prompt its request actually carried; the sample
|
||||
// must anchor against the surface as of that request, not after the
|
||||
// assistant message joins it.
|
||||
const answer = appendAssistant(session, 'an answer of some length', { inputTokens: 900, outputTokens: 20 }, 1, 1)
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
const afterTurn = pressure(ctx, session)
|
||||
expect(afterTurn.pressureTokens).toBe(900)
|
||||
// The assistant message landed after the sample, so it already shows.
|
||||
expect(afterTurn.projectedTokens).toBeGreaterThan(900)
|
||||
|
||||
const grown = appendUser(session, 'a follow-up question that grows the surface further')
|
||||
const beforeCompaction = pressure(ctx, session).projectedTokens
|
||||
expect(beforeCompaction).toBeGreaterThan(afterTurn.projectedTokens!)
|
||||
|
||||
// Compaction reports no usage of its own, so `pressureTokens` cannot move;
|
||||
// the projected figure must shrink anyway — the defect this field fixes.
|
||||
appendSummaryMeter(ctx, session, question, grown)
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'summary' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
}), {
|
||||
surfaceOp: { op: 'replace', start: question, end: grown },
|
||||
sourceEventSeqs: [question, answer, grown],
|
||||
})
|
||||
const compacted = pressure(ctx, session)
|
||||
expect(compacted.pressureTokens).toBe(900)
|
||||
expect(compacted.projectedTokens).toBeLessThan(beforeCompaction!)
|
||||
})
|
||||
|
||||
it('folds a replacement without a claim at zero', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
const question = appendUser(session, 'a question from an unmetered log')
|
||||
startStep(session, 1, 1)
|
||||
usageChunk(session, { inputTokens: 100, outputTokens: 1 }, 1, 1)
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
const before = pressure(ctx, session)
|
||||
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: 'summary without a preceding claim' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
}), {
|
||||
surfaceOp: { op: 'replace', start: question, end: question },
|
||||
sourceEventSeqs: [question],
|
||||
})
|
||||
|
||||
expect(pressure(ctx, session)).toEqual(before)
|
||||
})
|
||||
|
||||
it('clamps a projection that heuristic error drove below zero', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
recordContext(session, 'large', 128_000)
|
||||
const question = appendUser(session, 'a question long enough to outprice the sample'.repeat(4))
|
||||
startStep(session, 1, 1)
|
||||
// A provider sample far below the heuristic price of what it replaced:
|
||||
// shadowing that span subtracts more than the sample holds.
|
||||
appendAssistant(session, 'ok', { inputTokens: 3, outputTokens: 1 }, 1, 1)
|
||||
session.append('step/end', { turn: 1, step: 1 })
|
||||
appendSummaryMeter(ctx, session, question, question)
|
||||
session.append('user/message', createUserMessage({
|
||||
content: [{ type: 'text', text: '.' }],
|
||||
source: { kind: 'plugin', plugin: 'test' },
|
||||
}), {
|
||||
surfaceOp: { op: 'replace', start: question, end: question },
|
||||
sourceEventSeqs: [question],
|
||||
})
|
||||
expect(pressure(ctx, session).projectedTokens).toBe(0)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -23,6 +23,9 @@
|
||||
{
|
||||
"path": "../../core/session"
|
||||
},
|
||||
{
|
||||
"path": "../../compact/compact"
|
||||
},
|
||||
{
|
||||
"path": "../../session-projection/session-projection"
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user