From 4c80cab108166d47f2729c22dcb9298179194361 Mon Sep 17 00:00:00 2001 From: Yichen Jiang Date: Tue, 4 Aug 2026 11:42:35 +0800 Subject: [PATCH] fix(llm): capture an immutable snapshot per pi-ai operation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review found four defects in the declared-provider work. `PiAiAdapter` reused one `Models` collection and mutated it whenever the configuration changed. `Models.streamSimple()` resolves its provider lazily — when the stream is first consumed, which is after the adapter awaits the route's credential — so a configuration change landing in that window let an in-flight request finish under a configuration it never resolved against, or fail on a provider that no longer existed. Each resolution now produces an immutable snapshot and every operation captures one before its first await, which is what makes the seam's per-step freeze (`llm.prepareCall()`) hold end to end: switching models mid-reply takes effect on the next step, never inside the one in flight. `defaultMaxTokens` was materialized from the catalog's `Model.maxTokens`. The two answer different questions: pi-ai requires that field as the model's output capability, while the seam's is a cap the deployment chose to send on requests naming none, so every request had started carrying a number nobody picked. Only an explicitly configured cap reaches the seam now. The configurable-provider directory was refreshed by disposing its registration and making a new one. A candidate set the registry refuses — a profile keyed `deepseek-official`, which llm-deepseek declares — left the whole directory withdrawn and the Models page empty, silently, because the settings callback contains the failure. The seam's registration handle now carries `replace()` with the same validate-first atomicity `registerAdapter` has. The protocol table offered every pi-ai streaming API, including four whose authentication a profile cannot express: Bedrock signs with SigV4 over AWS credentials and a region, Vertex needs a project, a location, and ADC, Azure needs provider environment plus an api-version, and Codex uses OAuth. Offering them handed back routes that cannot authenticate. Catalog routes still reach them through their own provider. --- ...-pi-ai-declared-provider-catalog.i18n.yaml | 4 +- ...6-08-03-pi-ai-declared-provider-catalog.md | 20 +- ...8-03-pi-ai-declared-provider-catalog.zh.md | 20 +- docs/config-catalog.md | 8 +- docs/cordis-catalog/services.md | 8 +- .../cordis/tool-cordis/src/api-catalog.ts | 8 +- packages/llm/llm-pi-ai/README.i18n.yaml | 4 +- packages/llm/llm-pi-ai/README.md | 8 +- packages/llm/llm-pi-ai/README.zh.md | 8 +- packages/llm/llm-pi-ai/src/adapter.ts | 95 +++++---- packages/llm/llm-pi-ai/src/catalog.ts | 34 ++- packages/llm/llm-pi-ai/src/config.ts | 11 +- packages/llm/llm-pi-ai/src/index.ts | 28 ++- packages/llm/llm-pi-ai/src/provider.ts | 24 ++- packages/llm/llm-pi-ai/tests/catalog.spec.ts | 197 +++++++++++++++++- packages/llm/llm/README.i18n.yaml | 4 +- packages/llm/llm/README.md | 2 +- packages/llm/llm/README.zh.md | 2 +- packages/llm/llm/src/index.ts | 68 +++++- packages/llm/llm/tests/topology.spec.ts | 26 +++ scripts/gen-cordis-catalog.ts | 1 + 21 files changed, 476 insertions(+), 104 deletions(-) diff --git a/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.i18n.yaml b/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.i18n.yaml index 7fb32d2c29..738dc42275 100644 --- a/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.md -2026-08-03-pi-ai-declared-provider-catalog.md: ef695f6e4c79725400ee39a2d40ead27d6559a8d -2026-08-03-pi-ai-declared-provider-catalog.zh.md: 13cfed574f646de07228da80b3e518e31fd1f50b +2026-08-03-pi-ai-declared-provider-catalog.md: 343b54c5a09be17c0e4c31c219c01b0f1119847a +2026-08-03-pi-ai-declared-provider-catalog.zh.md: 8bc1fceb165cf6477be973944f0091db9bbe75f3 diff --git a/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.md b/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.md index ef695f6e4c..343b54c5a0 100644 --- a/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.md +++ b/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.md @@ -15,8 +15,17 @@ The adapter also streamed through `streamSimple` from `@earendil-works/pi-ai/com A provider route is a **declaration**, and the installed catalog is its default. `resolveProfiles` no longer checks route keys against `getBuiltinProviders()`. Instead each route resolves to a materialized model list plus the pi-ai `Provider` that serves it: - `catalog.ts` merges the installed catalog under the profile's own entries. A profile's `models` list *replaces* the route's catalog (an absent or empty list serves it unchanged), and each entry defaults its unset fields from the installed model of the same `id`. Only the fields the harness consumes are configurable — `id`, `name`, `contextWindow`, `maxTokens`, `reasoning`. Pricing and input modalities are absent from the surface because nothing reads them: `replay.ts` zeroes pi-ai's cost metadata and `context.ts` keeps only text blocks. Reasoning-level spellings, OpenAI-compatibility quirks, and model headers ride the installed entry, because restating them in configuration could not be validated. -- `provider.ts` builds the route's `Provider`. A catalog route that keeps its catalog protocol **reuses** the installed provider with `getModels()` replaced; every other route is built by `createProvider()` over a protocol table whose entries are the same `@earendil-works/pi-ai/api/*.lazy` factories pi-ai's own provider factories use. -- `adapter.ts` owns one `createModels()` collection, re-synced when resolution produces a new profile map, and serves `listModels`, `resolveModel`, and `stream` from it. A model's configured `maxTokens` becomes the seam's `defaultMaxTokens`, so a request naming no output cap now carries the configured one. +- `provider.ts` builds the route's `Provider`. A catalog route that keeps its catalog protocol **reuses** the installed provider with `getModels()` replaced; every other route is built by `createProvider()` over a protocol table whose entries are the same `@earendil-works/pi-ai/api/*.lazy` factories pi-ai's own provider factories use. That table is narrower than pi-ai's full API set on purpose — it holds only protocols a profile can completely describe with a key, an endpoint, and headers, so Bedrock (SigV4 plus a region), Vertex (project, location, ADC), Azure (provider environment plus an api-version), and Codex (OAuth) are absent rather than offered as routes that cannot authenticate. Catalog routes still reach them through their own provider; only an explicit override is refused. +- `adapter.ts` turns each resolution into an **immutable snapshot** — the profiles plus a `createModels()` collection holding those providers — and every operation captures a whole snapshot before its first `await`. +- A model's **explicitly configured** `maxTokens` becomes the seam's `defaultMaxTokens`. The value inherited from the installed catalog does not: pi-ai requires `Model.maxTokens` as the model's output *capability*, while `defaultMaxTokens` is a cap the deployment chose to send on requests that name none, and materializing the former as the latter would start capping every request at a number nobody picked. + +### Snapshots, not a shared collection + +`Models.streamSimple()` resolves its provider lazily, when the returned stream is first consumed — which is after the adapter has awaited the route's credential. A single collection mutated in place would therefore let a request that started under one configuration finish under another, or fail on a provider that no longer exists, even though `llm.prepareCall()` already froze that step's config and captured its adapter registration. A configuration change builds a *new* collection and leaves the one in use alone, so the seam's per-step freeze holds all the way down: switching models mid-reply takes effect on the next step, never inside the one in flight. + +### The directory replaces atomically + +The configurable-provider directory follows the profiles, so it changes whenever a declared route appears or leaves. Withdrawing the old registration and making a new one cannot express that: a candidate set the registry refuses — a profile keyed `deepseek-official`, which `llm-deepseek` already declares — would leave this plugin's whole directory withdrawn and the Models page empty, silently, because the settings-change callback contains the failure. `registerConfigurableProviders` therefore returns a handle carrying `replace(entries)` with the same validate-the-candidate-set-first atomicity `registerAdapter` has, and the plugin uses it. A refused swap costs a diagnostic; the previous entries keep serving. Resolution fails loud and names the route and model at fault: a model the catalog does not describe needs an explicit `contextWindow` and `maxTokens`; a route the catalog does not ship needs `api`, `baseURL`, and a non-empty `models` list. Because the built `Provider` is part of the resolution result, a protocol or model error keeps the last good route set serving, exactly as a bad settings snapshot already did. @@ -34,14 +43,17 @@ pi-ai's `Models` carries its own credential concept — a `CredentialStore` keye - **Reuse the installed provider for catalog routes and `createProvider()` only for declared ones**, with no shared resolution. Zero risk to catalog behavior, but catalog materialization, endpoint override, and per-model configuration would each exist twice, and a catalog route that repoints its protocol would have to jump paths mid-resolution. The chosen split confines the asymmetry to provider construction, where it is forced by pi-ai not exposing a built provider's API implementations. - **Rebuild every route through `createProvider()`**, including catalog ones. Fully symmetric, but a built `Provider` does not expose its `api`, so the protocol table would become the ceiling on which providers work — Bedrock loads its Smithy module through a separate entry point and would silently stop working. - **Expose pi-ai's whole `Model` shape** (cost, input modalities, `thinkingLevelMap`, `compat`). Maximum configurability, but no current consumer reads those fields, so a configured price or modality would change nothing while reading as supported. + +- **Keep one mutable `Models` collection and re-sync it.** Fewer allocations, and correct for every operation that resolves synchronously. It is exactly wrong for the one that does not: `stream()` awaits a credential between capturing its model and dispatching it. +- **Simulate an atomic directory swap with dispose-then-register.** No seam change, and it works whenever the new set is valid — which is the case that never needed atomicity. - **A runtime dynamic catalog** — `fetchModels` plus `ModelsStore`, refreshed in the background. Rejected for this change: it makes the model list external mutable state needing cache, invalidation, and an offline path, and the product need is a one-shot discovery action whose result the user adopts into `settings.yaml`. That action belongs to the configuration surface and is deferred with it; `settings.yaml` stays the single source of truth for what a route serves. ## Consequences -Configuring a provider no longer depends on a pi-ai release. A gateway, a self-hosted server, or a model newer than the pinned catalog is a `settings.yaml` edit, and a stale context window can be corrected in place. The deprecated `/compat` import is gone, so pi-ai deleting it is no longer a breaking event. `defaultMaxTokens` now flows from configuration, closing the case where a request carried no output cap at all. +Configuring a provider no longer depends on a pi-ai release. A gateway, a self-hosted server, or a model newer than the pinned catalog is a `settings.yaml` edit, and a stale context window can be corrected in place. The deprecated `/compat` import is gone, so pi-ai deleting it is no longer a breaking event. `defaultMaxTokens` now flows from configuration when a deployment states one, without inventing a cap from catalog metadata. What it costs: `settings.yaml` grows for a declared route, because a model the catalog cannot default must state its own capacity. `api` applies to a whole route, so a mixed-protocol catalog route cannot host a model of the other protocol — splitting it across two route keys is the workaround. Nothing queries a provider's `/models`, so a model list is only as current as its last edit. Reported error shape shifts in one case: a route whose auth resolves to nothing now surfaces pi-ai's own diagnostic as an error `finish` chunk before any network call, where the previous adapter sent a keyless request and surfaced the provider's 401. ## Testing -`tests/catalog.spec.ts` covers the contract end to end against local mock servers: a hand-declared route streaming to its own endpoint with its own credential, its appearance in the configurable-provider directory, per-model overrides defaulting from the installed catalog, a model added to a catalog route, protocol repointing with and without an endpoint override, catalog-only metadata surviving an override, the keyless posture and its `Authorization`-header workaround, and every resolution failure that names a route or model. `tests/sdk-options.spec.ts` re-targets the SDK boundary from the removed `/compat` import to the protocol table's lazy api module, which also pins that a setup failure arrives as a terminal error chunk rather than a throw. The twin's [design-verification role](2026-06-13-twin-llm-adapters.md) is unchanged. +`tests/catalog.spec.ts` covers the contract end to end against local mock servers: a hand-declared route streaming to its own endpoint with its own credential, its appearance in the configurable-provider directory, per-model overrides defaulting from the installed catalog, a model added to a catalog route, protocol repointing with and without an endpoint override, catalog-only metadata surviving an override, the keyless posture and its `Authorization`-header workaround, and every resolution failure that names a route or model. `tests/catalog.spec.ts` also pins the snapshot and directory contracts: an in-flight request whose route set changes during its credential await still reaches the endpoint it resolved against, the next request picks up the new one, a colliding declared route leaves the directory whole, and a declared route's entry appears and leaves with its profile. `packages/llm/llm/tests/topology.spec.ts` covers `replace` — refusing a candidate another registration owns while keeping the current set, accepting a swap over its own entries, allowing an empty set, and failing after disposal. `tests/sdk-options.spec.ts` re-targets the SDK boundary from the removed `/compat` import to the protocol table's lazy api module, which also pins that a setup failure arrives as a terminal error chunk rather than a throw. The twin's [design-verification role](2026-06-13-twin-llm-adapters.md) is unchanged. diff --git a/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.zh.md b/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.zh.md index 13cfed574f..8bc1fceb16 100644 --- a/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.zh.md +++ b/.agents/notes/implemented/architecture/2026-08-03-pi-ai-declared-provider-catalog.zh.md @@ -15,8 +15,17 @@ Status: implemented 提供方路由是一份**声明**,已安装 catalog 是它的默认值。`resolveProfiles` 不再拿路由键去核对 `getBuiltinProviders()`,而是把每条路由解析成一份物化模型列表,外加服务它的 pi-ai `Provider`: - `catalog.ts` 把已安装 catalog 合并到 profile 自身条目之下。profile 的 `models` 列表*替换*该路由的 catalog(列表缺席或为空则原样服务),每个条目从同 `id` 的已安装模型继承自身未设置的字段。只有 harness 会消费的字段可配置——`id`、`name`、`contextWindow`、`maxTokens`、`reasoning`。定价与输入模态不出现在配置面,因为没有任何读取方:`replay.ts` 把 pi-ai 的成本元数据清零,`context.ts` 只保留文本块。思考级别拼写、OpenAI 兼容性怪癖与模型标头沿用已安装条目,因为在配置里重述它们无法被校验。 -- `provider.ts` 构造路由的 `Provider`。保持 catalog 协议不变的 catalog 路由会**复用**已安装提供方,只替换 `getModels()`;其余路由都由 `createProvider()` 基于一张协议表构造,表中条目正是 pi-ai 自己的提供方工厂所用的 `@earendil-works/pi-ai/api/*.lazy` factory。 -- `adapter.ts` 持有一个 `createModels()` 集合,在解析产出新的 profile 映射时重新同步,并由它服务 `listModels`、`resolveModel` 与 `stream`。模型已配置的 `maxTokens` 会成为 seam 的 `defaultMaxTokens`,因此未点名输出上限的请求现在会携带已配置的那一个。 +- `provider.ts` 构造路由的 `Provider`。保持 catalog 协议不变的 catalog 路由会**复用**已安装提供方,只替换 `getModels()`;其余路由都由 `createProvider()` 基于一张协议表构造,表中条目正是 pi-ai 自己的提供方工厂所用的 `@earendil-works/pi-ai/api/*.lazy` factory。该表刻意窄于 pi-ai 的完整 API 集合——只保留 profile 能用密钥、端点与标头完整描述的协议,因此 Bedrock(SigV4 加 region)、Vertex(project、location、ADC)、Azure(提供方环境加 api-version)与 Codex(OAuth)不在其中,而不是被当作无法认证的路由提供出去。catalog 路由仍可经自己的 provider 抵达它们;被拒的只有显式覆盖。 +- `adapter.ts` 把每次解析变成一份**不可变快照**——profiles 加上持有这些 provider 的 `createModels()` 集合——每个操作都在自己第一个 `await` 之前整体捕获一份。 +- 模型**显式配置**的 `maxTokens` 会成为 seam 的 `defaultMaxTokens`;从已安装 catalog 继承来的那份不会:pi-ai 要求 `Model.maxTokens` 表示模型的输出**能力**,而 `defaultMaxTokens` 是部署选定、发给未点名上限的请求的那个值,把前者物化成后者会让每个请求都被一个无人选择的数字封顶。 + +### 快照,而不是共享集合 + +`Models.streamSimple()` 惰性解析 provider——在返回的流首次被消费时,而那已在适配器 await 路由凭据之后。因此就地改动的单一集合,会让一个在旧配置下开始的请求在新配置下结束,或者撞上一个已不存在的 provider,尽管 `llm.prepareCall()` 早已冻结了该步的 config 并捕获了其适配器注册。配置变化改为构造**新**集合,正在被使用的那个原封不动,于是 seam 的每步冻结得以贯通到底:回复途中切换模型在下一步生效,绝不影响在途的那一步。 + +### 目录原子替换 + +可配置提供方目录跟随 profiles,因此每当一条声明路由出现或离开它都会变化。「撤销旧注册再新建一个」表达不了这件事:注册表拒绝的候选集合——比如一份键为 `deepseek-official` 的 profile,而 `llm-deepseek` 已声明了它——会让本插件的整个目录被撤走、Models 页变空,而且是静默的,因为 settings 变更回调把失败容住了。因此 `registerConfigurableProviders` 改为返回带 `replace(entries)` 的句柄,其「候选集先整体校验」的原子性与 `registerAdapter` 相同,插件改用它。被拒的替换只付出一条诊断;先前的条目继续服务。 解析失败得响亮,并点名出问题的路由与模型:catalog 未描述的模型需要显式的 `contextWindow` 与 `maxTokens`;catalog 未提供的路由需要 `api`、`baseURL` 和非空的 `models` 列表。由于构造出的 `Provider` 是解析结果的一部分,协议或模型出错时最后可用的路由集合会继续服务——与此前坏的 settings 快照的行为完全一致。 @@ -34,14 +43,17 @@ pi-ai 的 `Models` 自带一套凭据概念——按提供方 id 索引的 `Cred - **catalog 路由复用已安装提供方,只有声明式路由走 `createProvider()`**,且两者不共享解析。对 catalog 行为零风险,但 catalog 物化、端点覆盖与每模型配置这三件事都要各写两遍,而改指协议的 catalog 路由还得在解析中途跳到另一条路径。已采纳的拆法把不对称收敛在提供方构造这一处——那里的不对称是 pi-ai 不暴露已构造提供方的 API 实现所强加的。 - **让每条路由都经 `createProvider()` 重建**,包括 catalog 路由。完全对称,但已构造的 `Provider` 不暴露自己的 `api`,于是协议表会成为「哪些提供方能用」的天花板——Bedrock 经独立入口加载其 Smithy 模块,会因此静默失效。 - **完整暴露 pi-ai 的 `Model` 形状**(成本、输入模态、`thinkingLevelMap`、`compat`)。可配置性最大,但这些字段当前没有任何读取方,因此配了价格或模态什么也不会改变,却看起来像是受支持的。 + +- **保留单个可变 `Models` 集合并重新同步。** 分配更少,且对每个同步完成解析的操作都是正确的;唯独对那个不同步的操作恰恰是错的:`stream()` 会在捕获模型与派发模型之间 await 一次凭据。 +- **用「先 dispose 再注册」模拟目录原子替换。** 无需改 seam,且在新集合有效时确实可用——而那正是从不需要原子性的那种情形。 - **运行时动态 catalog**——`fetchModels` 加 `ModelsStore`,后台刷新。本次变更拒绝:它把模型列表变成需要缓存、失效与离线路径的外部可变状态,而产品需求是一次性的发现动作、其结果由用户采纳进 `settings.yaml`。该动作属于配置界面,与之一并暂缓;`settings.yaml` 始终是「路由服务什么」的唯一事实源。 ## Consequences -配置一个提供方不再取决于 pi-ai 的发布节奏。网关、自建服务,或比锁定 catalog 更新的模型,都是一次 `settings.yaml` 编辑,过期的上下文窗口也能就地更正。废弃的 `/compat` 导入已经消失,因此 pi-ai 删除它不再是破坏性事件。`defaultMaxTokens` 现在自配置流出,堵上了「请求完全不带输出上限」的情形。 +配置一个提供方不再取决于 pi-ai 的发布节奏。网关、自建服务,或比锁定 catalog 更新的模型,都是一次 `settings.yaml` 编辑,过期的上下文窗口也能就地更正。废弃的 `/compat` 导入已经消失,因此 pi-ai 删除它不再是破坏性事件。`defaultMaxTokens` 现在只在部署明确给出时才自配置流出,不会从 catalog 元数据里发明一个上限。 代价是:声明式路由会让 `settings.yaml` 变长,因为 catalog 无法默认的模型必须自报容量。`api` 作用于整条路由,因此混合协议的 catalog 路由无法承载另一种协议的模型——把它拆成两个路由键是变通办法。没有任何环节查询提供方的 `/models`,因此模型列表的新鲜度只到最近一次编辑为止。有一种情形下报错形状发生变化:auth 解析不出任何值的路由,现在会在任何网络调用之前把 pi-ai 自己的诊断作为错误 `finish` 分片呈现,而此前的适配器会发出无密钥请求并呈现提供方的 401。 ## Testing -`tests/catalog.spec.ts` 针对本地 mock 服务器端到端覆盖该契约:手工声明的路由带着自己的凭据流向自己的端点、它在可配置提供方目录中的出现、每模型覆盖从已安装 catalog 继承默认值、向 catalog 路由添加模型、带与不带端点覆盖的协议改指、catalog 独有元数据在覆盖后存活、无密钥姿态及其 `Authorization` 标头变通,以及每一种点名路由或模型的解析失败。`tests/sdk-options.spec.ts` 把 SDK 边界从已移除的 `/compat` 导入改指到协议表的 lazy api 模块,同时钉住「setup 失败以终止性错误分片而非抛出的形式抵达」。twin 的[设计验证角色](2026-06-13-twin-llm-adapters.md)不变。 +`tests/catalog.spec.ts` 针对本地 mock 服务器端到端覆盖该契约:手工声明的路由带着自己的凭据流向自己的端点、它在可配置提供方目录中的出现、每模型覆盖从已安装 catalog 继承默认值、向 catalog 路由添加模型、带与不带端点覆盖的协议改指、catalog 独有元数据在覆盖后存活、无密钥姿态及其 `Authorization` 标头变通,以及每一种点名路由或模型的解析失败。`tests/catalog.spec.ts` 还钉住了快照与目录两项契约:在途请求即便其路由集在 credential await 期间改变,仍抵达它解析时对应的端点;下一个请求取用新配置;冲突的声明路由让目录保持完好;声明路由的条目随其 profile 出现与离开。`packages/llm/llm/tests/topology.spec.ts` 覆盖 `replace`——拒绝他人已拥有的候选同时保住当前集合、接受对自身条目的替换、允许空集合,以及 dispose 之后失败。`tests/sdk-options.spec.ts` 把 SDK 边界从已移除的 `/compat` 导入改指到协议表的 lazy api 模块,同时钉住「setup 失败以终止性错误分片而非抛出的形式抵达」。twin 的[设计验证角色](2026-06-13-twin-llm-adapters.md)不变。 diff --git a/docs/config-catalog.md b/docs/config-catalog.md index 44b54620d8..65ba351476 100644 --- a/docs/config-catalog.md +++ b/docs/config-catalog.md @@ -746,7 +746,11 @@ export interface PiAiModelProfile { name?: string /** Maximum combined request and response context in tokens. */ contextWindow?: number - /** Per-request output cap materialized when a caller omits one. */ + /** + * Maximum output tokens. Configuring one also makes it this model's + * per-request default; the value inherited from the installed catalog is the + * model's capability and never becomes a request default on its own. + */ maxTokens?: number /** Whether the model exposes reasoning; defaults to the catalog capability. */ reasoning?: boolean @@ -755,7 +759,7 @@ export interface PiAiModelProfile { Depends on: `CacheRetention` (`@earendil-works/pi-ai`) · `ModelThinkingLevel` (`@earendil-works/pi-ai`) · [`RetryPolicyConfig`](../packages/llm/llm/src/index.ts) · `ThinkingBudgets` (`@earendil-works/pi-ai`) · `Transport` (`@earendil-works/pi-ai`) -Source: [`packages/llm/llm-pi-ai/src/config.ts:98`](../packages/llm/llm-pi-ai/src/config.ts) +Source: [`packages/llm/llm-pi-ai/src/config.ts:104`](../packages/llm/llm-pi-ai/src/config.ts) ## `@deepseek-ai/dsh-llm-replay` diff --git a/docs/cordis-catalog/services.md b/docs/cordis-catalog/services.md index 36446f39c3..55173f08d2 100644 --- a/docs/cordis-catalog/services.md +++ b/docs/cordis-catalog/services.md @@ -834,9 +834,9 @@ listProviders(): LlmProviderInfo[] * entry, or a provider already declared by any registration throws * `LlmError` without registering the rest. Disposed with the fiber. * @param entries - every configurable provider this plugin owns. - * @returns the disposer that withdraws all of them. + * @returns a handle that withdraws all of them, and can atomically replace them. */ -registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void +registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle /** * List every declared configurable provider, registered or dormant. @@ -908,9 +908,9 @@ async prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise ``` -Types: [AdapterRegistrationHandle](../core-data-structures/core.md) · [GenerateOptions](../core-data-structures/core.md) · [LlmAdapter](../core-data-structures/llm-streaming.md) · [LlmCallConfig](../core-data-structures/core.md) · [LlmConfigurableProvider](../core-data-structures/core.md) · [LlmModelInfo](../core-data-structures/core.md) · [LlmProviderInfo](../core-data-structures/core.md) · [LlmResolvedModelInfo](../core-data-structures/core.md) · [PreparedLlmCall](../core-data-structures/llm-streaming.md) · [ResolvedRetryPolicy](../core-data-structures/llm-streaming.md) · [StreamChunk](../core-data-structures/llm-streaming.md) +Types: [AdapterRegistrationHandle](../core-data-structures/core.md) · [DirectoryRegistrationHandle](../core-data-structures/core.md) · [GenerateOptions](../core-data-structures/core.md) · [LlmAdapter](../core-data-structures/llm-streaming.md) · [LlmCallConfig](../core-data-structures/core.md) · [LlmConfigurableProvider](../core-data-structures/core.md) · [LlmModelInfo](../core-data-structures/core.md) · [LlmProviderInfo](../core-data-structures/core.md) · [LlmResolvedModelInfo](../core-data-structures/core.md) · [PreparedLlmCall](../core-data-structures/llm-streaming.md) · [ResolvedRetryPolicy](../core-data-structures/llm-streaming.md) · [StreamChunk](../core-data-structures/llm-streaming.md) -Source: [`packages/llm/llm/src/index.ts:232`](../../packages/llm/llm/src/index.ts) +Source: [`packages/llm/llm/src/index.ts:253`](../../packages/llm/llm/src/index.ts) ## `ctx.permission` — `PermissionService` diff --git a/packages/cordis/tool-cordis/src/api-catalog.ts b/packages/cordis/tool-cordis/src/api-catalog.ts index a5903d8be0..298a493730 100644 --- a/packages/cordis/tool-cordis/src/api-catalog.ts +++ b/packages/cordis/tool-cordis/src/api-catalog.ts @@ -421,8 +421,8 @@ export const SERVICE_API: readonly ServiceApiEntry[] = [ jsDoc: '/**\n * Describe provider routes with a registered adapter.\n * @returns detached provider metadata in registration order.\n */', }, { - signature: 'registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void', - jsDoc: '/**\n * Declare provider routes an adapter plugin can activate through\n * configuration. Registration is all-or-nothing: an empty list, invalid\n * entry, or a provider already declared by any registration throws\n * `LlmError` without registering the rest. Disposed with the fiber.\n * @param entries - every configurable provider this plugin owns.\n * @returns the disposer that withdraws all of them.\n */', + signature: 'registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle', + jsDoc: '/**\n * Declare provider routes an adapter plugin can activate through\n * configuration. Registration is all-or-nothing: an empty list, invalid\n * entry, or a provider already declared by any registration throws\n * `LlmError` without registering the rest. Disposed with the fiber.\n * @param entries - every configurable provider this plugin owns.\n * @returns a handle that withdraws all of them, and can atomically replace them.\n */', }, { signature: 'listConfigurableProviders(): LlmConfigurableProvider[]', @@ -1889,6 +1889,10 @@ export const TYPE_API: readonly TypeApiEntry[] = [ name: 'DirectoryPickerNativeCapability', declaration: 'export interface DirectoryPickerNativeCapability {\n kind: \'native\';\n pick(signal: AbortSignal): Promise;\n}', }, + { + name: 'DirectoryRegistrationHandle', + declaration: 'export interface DirectoryRegistrationHandle {\n (): void;\n replace(entries: readonly LlmConfigurableProvider[]): void;\n}', + }, { name: 'Domain', declaration: 'export interface Domain {\n readonly name: string;\n readonly global: DomainGlobalHandleOf;\n table(name: N): KvTable, TableValueOf>;\n close(): Promise;\n}', diff --git a/packages/llm/llm-pi-ai/README.i18n.yaml b/packages/llm/llm-pi-ai/README.i18n.yaml index 8f3d239ebf..5e87c87fb4 100644 --- a/packages/llm/llm-pi-ai/README.i18n.yaml +++ b/packages/llm/llm-pi-ai/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/llm/llm-pi-ai/README.md -README.md: e597eedeb4d6e0ebf402b5547f71c9aff370d3dd -README.zh.md: e28105f1253c138b9bb0baf5d00e0c7eba0d7b52 +README.md: dde6ce989a0fd87bf2dc60ce0d85deb1856d92d5 +README.zh.md: bb87ada7b5cf883b458c8cf9ace66bbd64fa154f diff --git a/packages/llm/llm-pi-ai/README.md b/packages/llm/llm-pi-ai/README.md index 78d32b0555..e3d8f04b9b 100644 --- a/packages/llm/llm-pi-ai/README.md +++ b/packages/llm/llm-pi-ai/README.md @@ -55,17 +55,19 @@ The dict shape makes duplicate routes unrepresentable, and the pre-release array A profile's `models` list *replaces* the route's installed catalog rather than extending it; omitting it (or leaving it empty) serves that catalog unchanged. Each entry defaults its unset fields from the installed model of the same `id`, so narrowing a catalog route to two models, correcting one capacity, or adding a model newer than the installed catalog are all one-line edits. Only the fields the harness consumes are configurable — `id`, `name`, `contextWindow`, `maxTokens`, and `reasoning`; pricing and input modalities have no harness consumer and ride the installed entry or are absent, while reasoning-level spellings and OpenAI-compatibility quirks have no configuration surface at all because restating them cannot be validated. -Resolution fails loud, naming the offending route and model, when a route cannot be served: a model the installed catalog does not describe needs an explicit `contextWindow` and `maxTokens`, and a route the catalog does not ship needs `api`, `baseURL`, and a non-empty `models` list. `api` accepts the protocols in `supportedProtocols()` — pi-ai's own streaming API set — and is only needed when the catalog cannot supply one: a model absent from the catalog inherits the protocol its shipped siblings agree on, so adding a model to a single-protocol catalog route restates nothing. +Resolution fails loud, naming the offending route and model, when a route cannot be served: a model the installed catalog does not describe needs an explicit `contextWindow` and `maxTokens`, and a route the catalog does not ship needs `api`, `baseURL`, and a non-empty `models` list. `api` accepts the protocols in `supportedProtocols()` and is only needed when the catalog cannot supply one: a model absent from the catalog inherits the protocol its shipped siblings agree on, so adding a model to a single-protocol catalog route restates nothing. `baseURL` sets the endpoint of every model on the route, so private proxies such as `https://proxy.example.com:8443` remain supported; a catalog route that omits it keeps each catalog model's own endpoint. Naming `api` on a catalog route repoints the whole route at that protocol, which is how a deployment moves a provider between, say, Responses and Chat Completions. +`supportedProtocols()` is deliberately narrower than pi-ai's full streaming API set: it holds only the protocols a profile can *completely* describe with a key, an endpoint, and headers. Bedrock signs with SigV4 over AWS credentials and a region, Vertex needs a project, a location, and application-default credentials, Azure needs provider environment plus an api-version, and Codex authenticates through OAuth — offering those would hand back a route that cannot authenticate. Catalog routes still reach them through their own provider; only an explicit override is refused. + ## Dynamic configuration (settings + credentials) The adapter reads its profiles through a thunk **once per operation** instead of freezing them at construction. The plugin registers the `llm-pi-ai` namespace on the optional `ctx.settings` seam with this same `Config` schema and its `cordis.yml` entry as the composition `base`, and because `providers` is a dict, the base and the user's `llm-pi-ai:` settings section merge **per provider**: a user can add a route, override one field of a composition route, or point a route at another proxy, all effective on the next request with no restart. Without a mounted settings service the entry config alone drives the adapter, unchanged. Credentials resolve per stream call: a non-empty literal `apiKey` wins, then `apiKeyEnv` through the optional `ctx.credentials` seam (`$DSH_HOME/.env` under the live environment; exactly that variable without a mounted seam). A profile naming no credential at all — and only that case — defers to pi-ai's ambient discovery. The route set and each route's captured retry policy are the registration-level facts: when either changes, the plugin replaces its registration atomically (same adapter instance, candidate set validated first), so a route another adapter already owns leaves the previous routes serving and reverting to a working configuration re-applies. Provider key order never counts as a change. A live settings snapshot naming an unknown provider (or failing any other resolver bound) keeps the last good profiles and logs the failure; the entry config itself still fails plugin load. -The adapter exposes each configured route's models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata read from the same pi-ai `Models` collection the request path uses, so discovery does not create a second model registry. `ctx.llm.resolveModelInfo(provider, model)` performs that exact descriptor lookup once and returns its identity, context window, configured output cap, and selectable thinking levels, keeping authoritative metadata on the route-owning adapter rather than its consumers. A model's `maxTokens` becomes the seam's `defaultMaxTokens`, so a request that names no output cap carries the configured one. +The adapter exposes each configured route's models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata read from the same pi-ai `Models` collection the request path uses, so discovery does not create a second model registry. `ctx.llm.resolveModelInfo(provider, model)` performs that exact descriptor lookup once and returns its identity, context window, configured output cap, and selectable thinking levels, keeping authoritative metadata on the route-owning adapter rather than its consumers. A model's **configured** `maxTokens` becomes the seam's `defaultMaxTokens`, so a request that names no output cap carries the one the deployment chose; a value inherited from the installed catalog is the model's output *capability* and never becomes a request default on its own. The `reasoning.efforts` list is pi-ai's ordered `getSupportedThinkingLevels(model)` result without filtering or normalization, including `off` and the model-specific availability of `xhigh` or `max`. The Harness exposes each canonical pi-ai level as an opaque ID; provider/model wire spellings remain inside pi-ai's `thinkingLevelMap`. A non-reasoning model therefore exposes pi-ai's `off` choice. The profile `reasoning` value, including `off`, is the deployment default when configured; omitting it preserves the provider default. Per-request `GenerateOptions.reasoningEffort` takes precedence, and any explicit value absent from the exact model capability fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O instead of being clamped. pi-ai's common stream options represent `off` by omitting `reasoning`. @@ -75,7 +77,7 @@ The adapter forces pi-ai's SDK `maxRetries` to zero so one `stream()` call makes ## Provider/model routing and replay -Each resolved route contributes one pi-ai `Provider` to the adapter's `createModels()` collection, and requests reach the provider through `Models.streamSimple()`. A catalog route that keeps its catalog protocol **reuses** the installed provider with its model list replaced, because that provider owns API implementations this package cannot reconstruct — Bedrock loads its Smithy module through a separate entry point — so rebuilding it from parts would silently narrow which providers work. Every other route is built by `createProvider()` over the protocol table behind `supportedProtocols()`, whose entries are the same factories pi-ai's own provider factories use. +Each resolution produces one **immutable** snapshot — the profiles plus a `createModels()` collection holding the `Provider` each route built — and every operation captures a whole snapshot before its first `await`. A configuration change builds a *new* collection rather than mutating the one in use: `Models.streamSimple()` resolves its provider lazily, when the stream is first consumed, which is after the credential await, so a mutated collection would let a request that started under one configuration finish under another or fail on a provider that no longer exists. This is what makes the seam's per-step call freeze (`llm.prepareCall()`) hold end to end — switching models mid-reply takes effect on the next step, never inside the one in flight. Requests reach their provider through `Models.streamSimple()`. A catalog route that keeps its catalog protocol **reuses** the installed provider with its model list replaced, because that provider owns API implementations this package cannot reconstruct — Bedrock loads its Smithy module through a separate entry point — so rebuilding it from parts would silently narrow which providers work. Every other route is built by `createProvider()` over the protocol table behind `supportedProtocols()`, whose entries are the same factories pi-ai's own provider factories use. Credentials never enter that collection. The harness resolves a route's key through its own seam before the request reaches pi-ai and passes it as the request's `apiKey` option, which pi-ai treats as the highest-priority auth override; `Models` therefore holds no credential store, and the harness keeps its fail-loud reference semantics. A route naming no credential resolves as configured-but-keyless and leaves the requirement to the protocol, which is where it actually lives. diff --git a/packages/llm/llm-pi-ai/README.zh.md b/packages/llm/llm-pi-ai/README.zh.md index dee44a81e3..98660943b7 100644 --- a/packages/llm/llm-pi-ai/README.zh.md +++ b/packages/llm/llm-pi-ai/README.zh.md @@ -55,17 +55,19 @@ profile 的 `models` 列表是*替换*该路由已安装 catalog,而不是扩充它;省略它(或留空)则原样服务该 catalog。每个条目都会从同 `id` 的已安装模型继承自身未设置的字段,因此把 catalog 路由收窄到两个模型、更正某个容量,或加入一个比已安装 catalog 更新的模型,都是一行编辑。只有 harness 会消费的字段可配置——`id`、`name`、`contextWindow`、`maxTokens` 与 `reasoning`;定价与输入模态没有 harness 消费方,因此沿用已安装条目或直接缺席,而思考级别的协议拼写与 OpenAI 兼容性怪癖则完全没有配置面,因为重述它们无法被校验。 -解析会失败得响亮,并点名出问题的路由与模型:已安装 catalog 未描述的模型需要显式的 `contextWindow` 与 `maxTokens`,catalog 未提供的路由则需要 `api`、`baseURL` 和非空的 `models` 列表。`api` 接受 `supportedProtocols()` 中的协议——即 pi-ai 自己的流式 API 集合——且仅在 catalog 无法提供协议时才需要:catalog 中不存在的模型会继承其同门模型一致同意的协议,因此向单协议 catalog 路由添加模型无需重述任何内容。 +解析会失败得响亮,并点名出问题的路由与模型:已安装 catalog 未描述的模型需要显式的 `contextWindow` 与 `maxTokens`,catalog 未提供的路由则需要 `api`、`baseURL` 和非空的 `models` 列表。`api` 接受 `supportedProtocols()` 中的协议,且仅在 catalog 无法提供协议时才需要:catalog 中不存在的模型会继承其同门模型一致同意的协议,因此向单协议 catalog 路由添加模型无需重述任何内容。 `baseURL` 设定该路由下每个模型的端点,因此仍支持 `https://proxy.example.com:8443` 等私有 proxy;省略它的 catalog 路由会保留每个 catalog 模型自己的端点。在 catalog 路由上点名 `api` 会把整条路由改指到该协议,这正是部署把某个提供方在 Responses 与 Chat Completions 之间迁移的方式。 +`supportedProtocols()` 刻意窄于 pi-ai 的完整流式 API 集合:它只保留 profile 能用密钥、端点与标头**完整描述**的那些协议。Bedrock 要用 AWS 凭据与 region 做 SigV4 签名,Vertex 需要 project、location 与应用默认凭据,Azure 需要提供方环境外加 api-version,Codex 走 OAuth——提供它们只会交回一个无法完成认证的路由。catalog 路由仍可经自己的 provider 抵达这些协议;被拒绝的只有显式覆盖。 + ## 动态配置(settings + credentials) 适配器经由一个 thunk **每操作读取一次** profile,而非在构造期冻结。插件在可选的 `ctx.settings` seam 上用同一份 `Config` schema 注册 `llm-pi-ai` namespace,并以其 `cordis.yml` 条目为组合 `base`;由于 `providers` 是字典,base 与用户的 `llm-pi-ai:` settings 分节**按提供方**合并:用户可以新增路由、覆盖组合路由的单个字段,或把路由指向另一个 proxy,全部在下一次请求生效,无需重启。未挂载 settings 服务时,仅由 entry 配置驱动适配器,行为不变。 凭据按每次 stream 调用解析:非空的字面 `apiKey` 优先,其次经可选的 `ctx.credentials` seam 解析 `apiKeyEnv`(活跃环境之下的 `$DSH_HOME/.env`;未挂载 seam 时恰好读取该环境变量)。只有完全没有点名任何凭据的 profile——仅限这一种情况——才交给 pi-ai 的环境发现。路由集合与每条路由捕获的重试策略是注册级事实:两者任一变化时,插件都会原子地替换自己的注册(同一适配器实例,候选集合先经校验),因此某条路由若已被另一适配器占有,先前的路由会继续服务,而改回可用配置时注册会重新生效。提供方键的顺序绝不算作变化。存活 settings 快照若点名未知提供方(或违反任何其他 resolver 约束),则保留最后可用 profile 并记录失败;entry 配置本身仍会使插件加载失败。 -适配器通过 `ctx.llm.listModels(provider)` 公开每条已配置路由的模型。这是从请求路径所用的同一个 pi-ai `Models` 集合读取的提供方无关 selector 元数据,因此发现不会创建第二个模型注册表。`ctx.llm.resolveModelInfo(provider, model)` 会执行一次精确 descriptor 查找,并返回其身份、上下文窗口、已配置输出上限和可选思考级别,让权威元数据保留在拥有路由的适配器上,而非消费方。模型的 `maxTokens` 会成为 seam 的 `defaultMaxTokens`,因此未点名输出上限的请求会携带已配置的那一个。 +适配器通过 `ctx.llm.listModels(provider)` 公开每条已配置路由的模型。这是从请求路径所用的同一个 pi-ai `Models` 集合读取的提供方无关 selector 元数据,因此发现不会创建第二个模型注册表。`ctx.llm.resolveModelInfo(provider, model)` 会执行一次精确 descriptor 查找,并返回其身份、上下文窗口、已配置输出上限和可选思考级别,让权威元数据保留在拥有路由的适配器上,而非消费方。模型**已配置**的 `maxTokens` 会成为 seam 的 `defaultMaxTokens`,因此未点名输出上限的请求会携带部署选定的那一个;而从已安装 catalog 继承来的值是模型的输出**能力**,绝不会自行变成请求默认值。 `reasoning.efforts` 列表是 pi-ai 有序的 `getSupportedThinkingLevels(model)` 结果,不经筛选或规范化,其中包括 `off`,以及模型对 `xhigh` 或 `max` 的特定支持。Harness 将每个规范 pi-ai 级别公开为不透明 ID;提供方/模型在协议格式中的表示仍保留在 pi-ai 的 `thinkingLevelMap` 中。因此,不具备推理(reasoning)能力的模型也会公开 pi-ai 的 `off` 选项。配置 profile 的 `reasoning` 值(包括 `off`)在存在时是部署默认值;省略它会保留提供方默认值。每次请求的 `GenerateOptions.reasoningEffort` 优先;任何未出现在确切模型能力中的显式值都会在网络 I/O 前以 `UNSUPPORTED_REASONING_EFFORT` 失败,而不会被自动调整。pi-ai 的通用流选项通过省略 `reasoning` 表示 `off`。 @@ -75,7 +77,7 @@ profile 的 `models` 列表是*替换*该路由已安装 catalog,而不是扩 ## 提供方/模型路由与回放 -每条已解析路由都会向适配器的 `createModels()` 集合贡献一个 pi-ai `Provider`,请求经 `Models.streamSimple()` 抵达提供方。保持 catalog 协议不变的 catalog 路由会**复用**已安装提供方,只替换其模型列表,因为该提供方持有本包无法重建的 API 实现——Bedrock 经由独立入口加载其 Smithy 模块——从零件重建会静默收窄可用提供方的范围。其余路由都由 `createProvider()` 基于 `supportedProtocols()` 背后的协议表构造,表中条目正是 pi-ai 自己的提供方工厂所用的同一批 factory。 +每次解析产出一份**不可变**快照——profiles 加上一个持有各路由所建 `Provider` 的 `createModels()` 集合——每个操作都在自己第一个 `await` 之前整体捕获一份快照。配置变化会构造**新**集合,而不是改动正在被使用的那个:`Models.streamSimple()` 是惰性的,它在流首次被消费时才解析 provider,而那已在 credential await 之后,因此改动共享集合会让一个在旧配置下开始的请求在新配置下结束,或者撞上一个已不存在的 provider。这正是 seam 的每步调用冻结(`llm.prepareCall()`)能贯通到底的原因——回复途中切换模型会在下一步生效,绝不会影响在途的那一步。请求经 `Models.streamSimple()` 抵达提供方。保持 catalog 协议不变的 catalog 路由会**复用**已安装提供方,只替换其模型列表,因为该提供方持有本包无法重建的 API 实现——Bedrock 经由独立入口加载其 Smithy 模块——从零件重建会静默收窄可用提供方的范围。其余路由都由 `createProvider()` 基于 `supportedProtocols()` 背后的协议表构造,表中条目正是 pi-ai 自己的提供方工厂所用的同一批 factory。 凭据绝不进入该集合。harness 在请求抵达 pi-ai 之前经自身 seam 解析路由密钥,并作为请求的 `apiKey` 选项传入,而 pi-ai 将其视为优先级最高的 auth 覆盖;因此 `Models` 不持有任何凭据存储,harness 也保住了自己失败得响亮的引用语义。没有点名任何凭据的路由会解析为「已配置但无密钥」,把该要求留给协议——那才是它真正所在的位置。 diff --git a/packages/llm/llm-pi-ai/src/adapter.ts b/packages/llm/llm-pi-ai/src/adapter.ts index f43bef2649..3a70d0f4af 100644 --- a/packages/llm/llm-pi-ai/src/adapter.ts +++ b/packages/llm/llm-pi-ai/src/adapter.ts @@ -1,10 +1,17 @@ /** * Generic pi-ai-backed implementation of the Harness LLM seam. * - * The adapter owns one pi-ai `Models` collection and keeps it in step with the - * resolved profiles: each route contributes the `Provider` its resolution built, - * so model lookup, protocol dispatch, and request auth all reach pi-ai through - * its supported runtime rather than the deprecated global compatibility entry. + * Each resolution produces one **immutable** snapshot — the profiles plus a + * `Models` collection holding the `Provider` each route built — and an + * operation captures a whole snapshot before its first `await`. A + * configuration change builds a *new* collection rather than mutating the one + * in use, because `Models.streamSimple()` is lazy: it resolves the provider + * when the stream is first consumed, which is after the credential await, so a + * mutated collection would let a request that started under one configuration + * finish under another — or fail with a provider that no longer exists. This is + * what makes the seam's per-step call freeze (`llm.prepareCall()`) hold all the + * way down: switching models mid-reply takes effect on the next step, never + * inside the one in flight. * * Credentials stay outside that collection. The harness resolves a route's key * through its own seam and passes it as the request's `apiKey` option, which @@ -18,6 +25,7 @@ import { createModels, getSupportedThinkingLevels } from '@earendil-works/pi-ai' import type { Api, Model, + Models, ModelThinkingLevel, MutableModels, SimpleStreamOptions, @@ -42,6 +50,14 @@ import type { ResolvedPiAiProviderProfile } from './config.ts' import { toPiContext } from './context.ts' import { toStreamChunks } from './stream.ts' +/** One resolution's frozen view: the profiles and the collection built from them. */ +interface PiAiSnapshot { + /** The resolved profiles this collection was built from, used as its identity. */ + profiles: ReadonlyMap + /** Providers for exactly those profiles; never mutated once published. */ + models: Models +} + /** Constructor options for {@link PiAiAdapter}: the two resolution seams the plugin owns. */ export interface PiAiAdapterOptions { /** Current validated profiles by provider route; called once per operation. */ @@ -107,41 +123,40 @@ function requestHeaders(headers: Readonly> | undefined): * restart; model descriptors come from the collection those profiles built. */ export class PiAiAdapter extends LlmAdapter { - private readonly models: MutableModels = createModels() - private registered: ReadonlyMap | undefined + private snapshot: PiAiSnapshot | undefined constructor(private readonly config: PiAiAdapterOptions) { super() } /** - * The `Models` collection for the current profiles. Resolution memoizes its - * result, so an unchanged configuration is recognized by identity and the - * collection is rebuilt only when the route set or any profile actually - * changes. + * The snapshot for the current profiles. Resolution memoizes its result, so + * an unchanged configuration is recognized by identity; a changed one gets a + * brand-new collection, leaving any snapshot an operation already captured + * untouched for as long as that operation holds it. */ - private collection(): MutableModels { + private current(): PiAiSnapshot { const profiles = this.config.profiles() - if (profiles === this.registered) return this.models - this.models.clearProviders() - for (const profile of profiles.values()) this.models.setProvider(profile.piProvider) - this.registered = profiles - return this.models + if (this.snapshot?.profiles === profiles) return this.snapshot + const models: MutableModels = createModels() + for (const profile of profiles.values()) models.setProvider(profile.piProvider) + this.snapshot = { profiles, models } + return this.snapshot } - /** The profile for one route, or the seam's own not-owned failure. */ - private profileOf(provider: string): ResolvedPiAiProviderProfile { - const profile = this.config.profiles().get(provider) + /** The profile for one route within one snapshot, or the not-owned failure. */ + private profileOf(snapshot: PiAiSnapshot, provider: string): ResolvedPiAiProviderProfile { + const profile = snapshot.profiles.get(provider) if (profile === undefined) { throw new LlmError(`pi-ai adapter does not own provider "${provider}"`, 'NO_ADAPTER') } return profile } - /** The configured descriptor for one exact route/model pair. */ - private modelOf(provider: string, model: string): Model { - this.profileOf(provider) - const resolved = this.collection().getModel(provider, model) + /** The configured descriptor for one exact route/model pair within one snapshot. */ + private modelOf(snapshot: PiAiSnapshot, provider: string, model: string): Model { + this.profileOf(snapshot, provider) + const resolved = snapshot.models.getModel(provider, model) if (resolved === undefined) { throw new LlmError(`pi-ai provider "${provider}" has no configured model "${model}"`, 'UNKNOWN_MODEL') } @@ -149,13 +164,14 @@ export class PiAiAdapter extends LlmAdapter { } override providerRetryPolicy(provider: string): ResolvedRetryPolicy | undefined { - return this.config.profiles().get(provider)?.retryPolicy + return this.current().profiles.get(provider)?.retryPolicy } override listModels(provider: string): Promise { return Promise.resolve().then(() => { - this.profileOf(provider) - return this.collection().getModels(provider).map(model => ({ + const snapshot = this.current() + this.profileOf(snapshot, provider) + return snapshot.models.getModels(provider).map(model => ({ provider, id: model.id, name: model.name, @@ -169,16 +185,20 @@ export class PiAiAdapter extends LlmAdapter { _signal?: AbortSignal, ): Promise { return Promise.resolve().then(() => { - const profile = this.profileOf(provider) - const resolvedModel = this.modelOf(provider, model) + const snapshot = this.current() + const profile = this.profileOf(snapshot, provider) + const resolvedModel = this.modelOf(snapshot, provider, model) const levels = getSupportedThinkingLevels(resolvedModel) const defaultLevel = resolveReasoningLevel(resolvedModel, profile.reasoning) + // Only a cap the deployment configured is a request default; the + // catalog's `maxTokens` sizes the model and stops there. + const configuredMaxTokens = profile.configuredMaxTokens.get(model) return { provider, id: model, name: resolvedModel.name, context: { contextWindow: resolvedModel.contextWindow }, - defaultMaxTokens: resolvedModel.maxTokens, + ...configuredMaxTokens === undefined ? {} : { defaultMaxTokens: configuredMaxTokens }, reasoning: { efforts: levels.map(level => ({ id: ReasoningEffortId(level), @@ -196,13 +216,14 @@ export class PiAiAdapter extends LlmAdapter { if (options.stop !== undefined) { throw new LlmError('llm-pi-ai does not support GenerateOptions.stop', 'UNSUPPORTED_OPTION') } - // One resolution per stream call: the profile snapshot, the model - // descriptor, and the credential freeze here and hold for this whole - // request, so an in-flight stream never observes a configuration change and - // the next call re-resolves. - const profile = this.profileOf(options.provider) - const collection = this.collection() - const model = this.modelOf(options.provider, options.model) + // One capture per stream call, taken before any await: the profile, the + // model descriptor, and the collection all come from the same immutable + // snapshot, and the credential freezes with them. A configuration change + // mid-request builds a separate snapshot, so this request finishes under + // the one it started with and the next call picks up the new one. + const snapshot = this.current() + const profile = this.profileOf(snapshot, options.provider) + const model = this.modelOf(snapshot, options.provider, options.model) const reasoning = resolveReasoningLevel( model, options.reasoningEffort ?? profile.reasoning, @@ -217,7 +238,7 @@ export class PiAiAdapter extends LlmAdapter { using watchdog = idleWatchdog(upstream, streamIdleTimeoutMs, 'LLM_STREAM_IDLE_TIMEOUT') try { - const events = collection.streamSimple(model, toPiContext(options), { + const events = snapshot.models.streamSimple(model, toPiContext(options), { ...profileOptions(profile, reasoning, apiKey), ...options.temperature === undefined ? {} : { temperature: options.temperature }, ...options.maxTokens === undefined ? {} : { maxTokens: options.maxTokens }, diff --git a/packages/llm/llm-pi-ai/src/catalog.ts b/packages/llm/llm-pi-ai/src/catalog.ts index 41e5527b39..a68c5fd9e8 100644 --- a/packages/llm/llm-pi-ai/src/catalog.ts +++ b/packages/llm/llm-pi-ai/src/catalog.ts @@ -79,7 +79,11 @@ export interface PiAiModelProfile { name?: string /** Maximum combined request and response context in tokens. */ contextWindow?: number - /** Per-request output cap materialized when a caller omits one. */ + /** + * Maximum output tokens. Configuring one also makes it this model's + * per-request default; the value inherited from the installed catalog is the + * model's capability and never becomes a request default on its own. + */ maxTokens?: number /** Whether the model exposes reasoning; defaults to the catalog capability. */ reasoning?: boolean @@ -116,15 +120,32 @@ function sharedCatalogApi(defaults: ReadonlyMap>): string | u return apis.size === 1 ? [...apis][0] : undefined } +/** One route's materialized catalog, plus the request caps its profile chose. */ +export interface RouteCatalog { + /** The materialized models in configuration order. */ + models: readonly Model[] + /** + * Per-request output caps this profile explicitly configured, by model id. + * + * Separate from `Model.maxTokens` because the two answer different + * questions: pi-ai requires `maxTokens` as the model's output *capability*, + * while the harness seam's `defaultMaxTokens` is a cap the deployment chose + * to send on requests that name none. Materializing a catalog capability as + * a request default would start capping every request at a number nobody + * picked, so only an explicit configuration lands here. + */ + configuredMaxTokens: ReadonlyMap +} + /** * Materialize one route's catalog by merging the installed catalog defaults * under the configured entries. A route with no configured `models` serves the * installed catalog unchanged, which is what keeps an existing * `providers: { deepseek: { apiKeyEnv: … } }` profile working untouched. * @param request - the route-level catalog facts. - * @returns the materialized models in configuration order. + * @returns the materialized models and the explicitly configured request caps. */ -export function resolveRouteModels(request: RouteCatalogRequest): readonly Model[] { +export function resolveRouteModels(request: RouteCatalogRequest): RouteCatalog { const { provider } = request const defaults = catalogModels(provider) const providerBaseUrl = catalogProvider(provider)?.baseUrl @@ -141,7 +162,8 @@ export function resolveRouteModels(request: RouteCatalogRequest): readonly Model } const routeApi = sharedCatalogApi(defaults) const seen = new Set() - return entries.map((entry) => { + const configuredMaxTokens = new Map() + const models = entries.map((entry) => { if (entry.id.length === 0) invalid(provider, 'has a model with an empty id') if (seen.has(entry.id)) invalid(provider, `lists model "${entry.id}" more than once`) seen.add(entry.id) @@ -171,6 +193,9 @@ export function resolveRouteModels(request: RouteCatalogRequest): readonly Model if (!Number.isInteger(maxTokens) || maxTokens <= 0) { invalid(provider, `model "${entry.id}" maxTokens must be a positive integer`) } + // Only a value the profile named is a deployment choice; the catalog's is + // the model's capability and stays out of request defaults. + if (entry.maxTokens !== undefined) configuredMaxTokens.set(entry.id, entry.maxTokens) return { id: entry.id, name: entry.name ?? base?.name ?? entry.id, @@ -190,4 +215,5 @@ export function resolveRouteModels(request: RouteCatalogRequest): readonly Model ...base?.headers === undefined ? {} : { headers: base.headers }, } }) + return { models, configuredMaxTokens } } diff --git a/packages/llm/llm-pi-ai/src/config.ts b/packages/llm/llm-pi-ai/src/config.ts index a1199ad471..01a63efb4a 100644 --- a/packages/llm/llm-pi-ai/src/config.ts +++ b/packages/llm/llm-pi-ai/src/config.ts @@ -92,6 +92,12 @@ export interface ResolvedPiAiProviderProfile * serving requests. */ piProvider: Provider + /** + * Per-request output caps this profile explicitly configured, by model id. + * The seam materializes one only into a request that names no cap of its + * own, so a catalog capability must not appear here. + */ + configuredMaxTokens: ReadonlyMap } /** Plugin configuration: the provider routes this instance owns. */ @@ -200,7 +206,7 @@ export function resolveProfiles( // always shown route keys, and a catalog route must not silently rename // itself on every configuration surface just because it gained a profile. const displayName = source.displayName ?? provider - const models = resolveRouteModels({ + const catalog = resolveRouteModels({ provider, ...source.api === undefined ? {} : { api: source.api }, ...source.baseURL === undefined ? {} : { baseURL: source.baseURL }, @@ -216,12 +222,13 @@ export function resolveProfiles( retryPolicy: resolveRetryPolicy(retryPolicy, `llm-pi-ai: provider "${provider}" retryPolicy`), ...rest.headers === undefined ? {} : { headers: { ...rest.headers } }, ...rest.thinkingBudgets === undefined ? {} : { thinkingBudgets: { ...rest.thinkingBudgets } }, + configuredMaxTokens: catalog.configuredMaxTokens, piProvider: buildProvider({ provider, displayName, ...source.api === undefined ? {} : { api: source.api }, ...source.baseURL === undefined ? {} : { baseURL: source.baseURL }, - models, + models: catalog.models, }), }) } diff --git a/packages/llm/llm-pi-ai/src/index.ts b/packages/llm/llm-pi-ai/src/index.ts index 70b1d52b42..bc0e77658c 100644 --- a/packages/llm/llm-pi-ai/src/index.ts +++ b/packages/llm/llm-pi-ai/src/index.ts @@ -44,7 +44,7 @@ import type { Context } from 'cordis' import { LlmError } from '@deepseek-ai/dsh-llm' -import type { AdapterRegistrationHandle, LlmConfigurableProvider } from '@deepseek-ai/dsh-llm' +import type { AdapterRegistrationHandle, DirectoryRegistrationHandle, LlmConfigurableProvider } from '@deepseek-ai/dsh-llm' import { deepEqualJson, installSettingsSection, settingsNamespace } from '@deepseek-ai/dsh-settings' import { PiAiAdapter } from './adapter.ts' import { catalogProviderIds } from './catalog.ts' @@ -151,13 +151,21 @@ export function apply(ctx: Context, config: Config): void { // mounts — dormant or not — so configuration surfaces can offer every // pi-ai provider before any route exists. Hand-declared routes join it as // profiles appear, and leave with them. - let directory: (() => void) | undefined + let directory: DirectoryRegistrationHandle | undefined let directoryFacts: unknown const ensureDirectory = (): void => { const entries = directoryEntries(profiles()) if (deepEqualJson(entries, directoryFacts)) return - directory?.() - directory = ctx.llm.registerConfigurableProviders(entries) + // Atomic replace, never dispose-then-register: a route another adapter + // family already declares (a profile keyed `deepseek-official`) would + // otherwise leave this plugin's whole directory withdrawn and the Models + // page empty. The candidate set is validated first, so a collision keeps + // the previous entries serving and only costs a diagnostic. + if (directory === undefined) { + directory = ctx.llm.registerConfigurableProviders(entries) + } else { + directory.replace(entries) + } directoryFacts = entries } ensureDirectory() @@ -199,8 +207,16 @@ export function apply(ctx: Context, config: Config): void { onChange: () => { ensureRegistrationFacts() // The directory follows the profiles the registry accepted, so a route - // that failed to register is not advertised as configurable. - ensureDirectory() + // that failed to register is not advertised as configurable. A refused + // directory swap is contained here for the same reason the registry's + // is: the previous entries keep serving, and `directoryFacts` stays put + // so returning to a working configuration re-applies. + try { + ensureDirectory() + } catch (error) { + ctx.logger.error('llm-pi-ai: keeping the previous configurable-provider directory after a refused update') + ctx.logger.error(error) + } }, }) } diff --git a/packages/llm/llm-pi-ai/src/provider.ts b/packages/llm/llm-pi-ai/src/provider.ts index fdcacd217c..07ec533919 100644 --- a/packages/llm/llm-pi-ai/src/provider.ts +++ b/packages/llm/llm-pi-ai/src/provider.ts @@ -22,12 +22,8 @@ import { createProvider } from '@earendil-works/pi-ai' import type { Api, ApiKeyAuth, Model, Provider, ProviderStreams } from '@earendil-works/pi-ai' import { anthropicMessagesApi } from '@earendil-works/pi-ai/api/anthropic-messages.lazy' -import { azureOpenAIResponsesApi } from '@earendil-works/pi-ai/api/azure-openai-responses.lazy' -import { bedrockConverseStreamApi } from '@earendil-works/pi-ai/api/bedrock-converse-stream.lazy' import { googleGenerativeAIApi } from '@earendil-works/pi-ai/api/google-generative-ai.lazy' -import { googleVertexApi } from '@earendil-works/pi-ai/api/google-vertex.lazy' import { mistralConversationsApi } from '@earendil-works/pi-ai/api/mistral-conversations.lazy' -import { openAICodexResponsesApi } from '@earendil-works/pi-ai/api/openai-codex-responses.lazy' import { openAICompletionsApi } from '@earendil-works/pi-ai/api/openai-completions.lazy' import { openAIResponsesApi } from '@earendil-works/pi-ai/api/openai-responses.lazy' import { piMessagesApi } from '@earendil-works/pi-ai/api/pi-messages.lazy' @@ -35,18 +31,24 @@ import { catalogProvider } from './catalog.ts' /** * Wire protocols a configured route may name, mapped to pi-ai's lazily loaded - * implementations. The table is pi-ai's own streaming API set: each entry is - * the factory that pi-ai's matching provider factory uses, so a hand-declared - * route reaches exactly the implementation a catalog route would. + * implementations. Each entry is the factory that pi-ai's matching provider + * factory uses, so a hand-declared route reaches exactly the implementation a + * catalog route would. + * + * The table is deliberately narrower than pi-ai's full streaming API set: it + * holds only the protocols a profile can *completely* describe with a key, an + * endpoint, and headers. Bedrock signs with SigV4 over AWS credentials and a + * region, Vertex needs a project, a location, and application-default + * credentials, Azure needs provider environment plus an api-version, and + * Codex authenticates through OAuth — none of which this configuration shape + * can express, so offering them would hand back a provider that cannot + * authenticate. Catalog routes still reach those protocols through their own + * provider; only an explicit override is refused. */ const PROTOCOLS: Readonly ProviderStreams>> = { 'anthropic-messages': anthropicMessagesApi, - 'azure-openai-responses': azureOpenAIResponsesApi, - 'bedrock-converse-stream': bedrockConverseStreamApi, 'google-generative-ai': googleGenerativeAIApi, - 'google-vertex': googleVertexApi, 'mistral-conversations': mistralConversationsApi, - 'openai-codex-responses': openAICodexResponsesApi, 'openai-completions': openAICompletionsApi, 'openai-responses': openAIResponsesApi, 'pi-messages': piMessagesApi, diff --git a/packages/llm/llm-pi-ai/tests/catalog.spec.ts b/packages/llm/llm-pi-ai/tests/catalog.spec.ts index d0f1725845..b2684c1997 100644 --- a/packages/llm/llm-pi-ai/tests/catalog.spec.ts +++ b/packages/llm/llm-pi-ai/tests/catalog.spec.ts @@ -1,14 +1,43 @@ +import { mkdtemp, rm, writeFile } from 'node:fs/promises' +import { tmpdir } from 'node:os' +import { join } from 'node:path' import { afterEach, describe, expect, it } from 'vitest' import { Context } from 'cordis' import LlmService, { createUserMessage } from '@deepseek-ai/dsh-llm' +import type { StreamChunk } from '@deepseek-ai/dsh-llm' +import SettingsLocal from '@deepseek-ai/dsh-settings-local' +import { settingsNamespace } from '@deepseek-ai/dsh-settings' import * as LlmPiAi from '@deepseek-ai/dsh-llm-pi-ai' +import { PiAiAdapter } from '@deepseek-ai/dsh-llm-pi-ai' import { getBuiltinModels } from '@earendil-works/pi-ai/providers/all' import { resolveProfiles } from '../src/config.ts' -import { buildProvider } from '../src/provider.ts' +import { buildProvider, supportedProtocols } from '../src/provider.ts' import { assemble } from './assemble.ts' import { closeMockServers, mockServer, textEvents } from './mock-server.ts' -afterEach(async () => { await closeMockServers() }) +const homes: string[] = [] + +afterEach(async () => { + await closeMockServers() + await Promise.all(homes.splice(0).map(dir => rm(dir, { recursive: true, force: true }))) +}) + +/** A throwaway $DSH_HOME with an empty settings document. */ +async function home(): Promise { + const dir = await mkdtemp(join(tmpdir(), 'dsh-pi-catalog-')) + homes.push(dir) + await writeFile(join(dir, 'settings.yaml'), '') + return dir +} + +/** The dormant composition plus a real settings service, as the product mounts it. */ +async function bootWithSettings(dir: string, config: LlmPiAi.Config): Promise { + const ctx = new Context() + await ctx.plugin(LlmService) + await ctx.plugin(SettingsLocal, { path: join(dir, 'settings.yaml'), watch: false }) + await ctx.plugin(LlmPiAi, config) + return ctx +} /** A complete hand-declared route: nothing about it exists in pi-ai's catalog. */ function gateway(baseURL: string, overrides: Record = {}): LlmPiAi.Config { @@ -107,6 +136,19 @@ describe('hand-declared providers', () => { })).toThrow(/needs a baseURL/) }) + it.each(['bedrock-converse-stream', 'google-vertex', 'azure-openai-responses', 'openai-codex-responses'])( + 'refuses %s, whose authentication a profile cannot express', + (api) => { + // These need SigV4 credentials and a region, a project plus ADC, provider + // environment and an api-version, or OAuth — none of which a key, an + // endpoint, and headers can carry, so a route naming one would be built + // unable to authenticate. + expect(supportedProtocols()).not.toContain(api) + expect(() => buildProvider({ provider: 'acme-gateway', displayName: 'Acme', api, models: [] })) + .toThrow(/cannot serve; supported protocols are/) + }, + ) + it('rejects a protocol this build cannot serve, and a route that names none', () => { const spec = { provider: 'acme-gateway', displayName: 'Acme Gateway', models: [] } expect(() => buildProvider({ ...spec, api: 'quantum-telepathy' })) @@ -206,14 +248,35 @@ describe('catalog routes with per-model configuration', () => { }) const info = await ctx.llm.resolveModelInfo('deepseek', catalogModel.id) - // The configured field wins; name and output cap still come from the catalog. + // The configured field wins and the name still comes from the catalog. The + // catalog's own output cap is the model's capability, not a cap anyone + // chose, so it must not arrive as the request default. expect(info.context).toEqual({ contextWindow: 4096 }) expect(info.name).toBe(catalogModel.name) - expect(info.defaultMaxTokens).toBe(catalogModel.maxTokens) + expect(info.defaultMaxTokens).toBeUndefined() // An explicit list replaces the catalog rather than adding to it. expect((await ctx.llm.listModels('deepseek')).map(model => model.id)).toEqual([catalogModel.id]) }) + it('materializes a request default only from a configured output cap', async () => { + const server = await mockServer([]) + const [catalogModel] = getBuiltinModels('deepseek') + if (catalogModel === undefined) throw new Error('the installed catalog ships no deepseek model') + const ctx = await harness({ + providers: { + deepseek: { + apiKey: 'k', + baseURL: server.url, + models: [{ id: catalogModel.id, maxTokens: 4096 }], + }, + }, + }) + + // Configuring the cap is the deployment choosing one, so it becomes the + // default the seam materializes into requests that name none. + expect((await ctx.llm.resolveModelInfo('deepseek', catalogModel.id)).defaultMaxTokens).toBe(4096) + }) + it('adds a model the installed catalog does not describe to a catalog route', async () => { const server = await mockServer([{ events: textEvents }]) const ctx = await harness({ @@ -262,6 +325,23 @@ describe('catalog routes with per-model configuration', () => { expect(model?.contextWindow).toBe(4096) }) + it('delegates both stream methods back to the reused catalog provider', async () => { + const server = await mockServer([{ events: textEvents }, { events: textEvents }]) + const resolved = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${server.url}/v1` } }) + const built = resolved.get('deepseek')?.piProvider + if (built === undefined) throw new Error('the deepseek route built no provider') + const [model] = built.getModels() + if (model === undefined) throw new Error('the deepseek route resolved no models') + const context = { messages: [{ role: 'user' as const, content: 'hi', timestamp: 0 }] } + + // `stream` is interface-required and unused by the harness adapter, which + // only calls `streamSimple`; both must still reach the catalog provider. + for await (const _event of built.stream(model, context, { apiKey: 'k' })) { /* drain */ } + for await (const _event of built.streamSimple(model, context, { apiKey: 'k' })) { /* drain */ } + + expect(server.paths).toEqual(['/v1/chat/completions', '/v1/chat/completions']) + }) + it('keeps each model its own endpoint when the catalog route declares none', () => { // `opencode` ships no provider-level endpoint: the address lives on every // catalog model, so the route resolves without any configured baseURL. @@ -300,3 +380,112 @@ describe('catalog routes with per-model configuration', () => { expect(server.paths).toEqual(['/v1/chat/completions']) }) }) + +describe('resolution snapshots', () => { + it('finishes an in-flight request under the configuration it started with', async () => { + const server = await mockServer([{ events: textEvents }]) + let current = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${server.url}/v1` } }) + let release: () => void = () => {} + const held = new Promise((resolve) => { release = resolve }) + const adapter = new PiAiAdapter({ + profiles: () => current, + // Credential resolution is the real await inside a stream call, and the + // window a configuration change has to land in. + resolveApiKey: async () => { await held; return 'k' }, + }) + + const chunks: StreamChunk[] = [] + const inFlight = (async () => { + for await (const chunk of adapter.stream({ + provider: 'deepseek', + model: 'deepseek-v4-flash', + messages: [], + })) chunks.push(chunk) + })() + + // The route set changes while the request waits, and something else reads + // the adapter meanwhile, which is what would rebuild a shared collection. + current = resolveProfiles({ openai: { apiKey: 'k', baseURL: `${server.url}/v1` } }) + await expect(adapter.listModels('openai')).resolves.not.toHaveLength(0) + release() + await inFlight + + // The in-flight request keeps its own snapshot: it reaches the endpoint it + // resolved against instead of failing on a provider that no longer exists. + expect(chunks.at(-1)).toMatchObject({ type: 'finish', reason: { kind: 'stop' } }) + expect(server.paths).toEqual(['/v1/chat/completions']) + }) + + it('serves the next request from the new configuration', async () => { + const first = await mockServer([{ events: textEvents }]) + const second = await mockServer([{ events: textEvents }]) + let current = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${first.url}/v1` } }) + const adapter = new PiAiAdapter({ profiles: () => current, resolveApiKey: () => Promise.resolve('k') }) + const drain = async (): Promise => { + for await (const _chunk of adapter.stream({ + provider: 'deepseek', model: 'deepseek-v4-flash', messages: [], + })) { /* drain */ } + } + + await drain() + current = resolveProfiles({ deepseek: { apiKey: 'k', baseURL: `${second.url}/v1` } }) + await drain() + + expect(first.paths).toHaveLength(1) + expect(second.paths).toHaveLength(1) + }) +}) + +describe('configurable-provider directory', () => { + it('keeps the previous directory when a route collides with another adapter family', async () => { + const dir = await home() + const ctx = await bootWithSettings(dir, {}) + // Another adapter family owns this route id, exactly as llm-deepseek does. + ctx.llm.registerConfigurableProviders([ + { provider: 'deepseek-official', displayName: 'DeepSeek', settingsNs: 'llm-deepseek', settingsPath: [] }, + ]) + const before = ctx.llm.listConfigurableProviders().length + expect(before).toBeGreaterThan(30) + + await ctx.settings.update(settingsNamespace('llm-pi-ai'), { + providers: { + 'deepseek-official': { + apiKey: 'k', + api: 'openai-completions', + baseURL: 'https://acme.test/v1', + models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }], + }, + }, + }) + + // The refused swap costs a diagnostic, not the directory: every entry the + // page needs is still declared. + expect(ctx.llm.listConfigurableProviders()).toHaveLength(before) + expect(ctx.llm.listConfigurableProviders().find(entry => entry.provider === 'deepseek-official')?.settingsNs) + .toBe('llm-deepseek') + }) + + it('replaces its entries atomically as declared routes come and go', async () => { + const dir = await home() + const ctx = await bootWithSettings(dir, {}) + const catalogOnly = ctx.llm.listConfigurableProviders().length + + await ctx.settings.update(settingsNamespace('llm-pi-ai'), { + providers: { + 'acme-gateway': { + apiKey: 'k', + displayName: 'Acme Gateway', + api: 'openai-completions', + baseURL: 'https://acme.test/v1', + models: [{ id: 'm', contextWindow: 1, maxTokens: 1 }], + }, + }, + }) + expect(ctx.llm.listConfigurableProviders()).toHaveLength(catalogOnly + 1) + expect(ctx.llm.listConfigurableProviders().find(entry => entry.provider === 'acme-gateway')?.displayName) + .toBe('Acme Gateway') + + await ctx.settings.replace(settingsNamespace('llm-pi-ai'), {}) + expect(ctx.llm.listConfigurableProviders()).toHaveLength(catalogOnly) + }) +}) diff --git a/packages/llm/llm/README.i18n.yaml b/packages/llm/llm/README.i18n.yaml index a561022e43..473187c895 100644 --- a/packages/llm/llm/README.i18n.yaml +++ b/packages/llm/llm/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/llm/llm/README.md -README.md: 21f428fb22c9a59a67d86f446ea866c1629b964a -README.zh.md: 9bd26993bc63b39de3d3a8039fea6b046c875504 +README.md: e09ec685ed0ab1e2492749237c277a874eb3b246 +README.zh.md: ca98e875a90eb16e32bc405d77cd5b2b56644180 diff --git a/packages/llm/llm/README.md b/packages/llm/llm/README.md index 21f428fb22..e09ec685ed 100644 --- a/packages/llm/llm/README.md +++ b/packages/llm/llm/README.md @@ -12,7 +12,7 @@ An adapter registry plus a single streaming call surface, interceptable via a wa - `ctx.llm.registerAdapter(providers: string[], adapter: LlmAdapter): AdapterRegistrationHandle` Register one adapter instance for the given provider routes. Registration is all-or-nothing, and is disposed with the calling fiber. The returned disposer also carries `replace(providers)`: the candidate route set is validated in full before anything moves, so a conflict with another adapter leaves the current routes registered and serving, and the swap itself is one synchronous section with no observable gap. `replace([])` is legal — a registration holding zero routes — unlike an empty initial registration. - `ctx.llm.listProviders(): LlmProviderInfo[]` Describe registered provider routes in registration order. -- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void` Declare provider routes an adapter plugin can activate through configuration — registered or dormant — each naming its owning settings namespace and the path to its profile inside that section. All-or-nothing (`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`), disposed with the calling fiber. +- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle` Declare provider routes an adapter plugin can activate through configuration — registered or dormant — each naming its owning settings namespace and the path to its profile inside that section. All-or-nothing (`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`), disposed with the calling fiber. The handle also carries `replace(entries)`: the candidate set is validated in full before anything moves, so an entry another registration already declares leaves the current set intact, and an empty array is legal there. A plugin whose declared set follows its configuration must use `replace` rather than disposing and re-registering — the latter strands the directory empty whenever the new set is refused. - `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` List the declared directory in declaration order; configuration surfaces merge it with `listProviders()` to mark each entry live or dormant. - `ctx.llm.providerRetryPolicy(provider: string): ResolvedRetryPolicy` Return the provider-owned retry policy captured during registration, with normal defaults resolved. - `ctx.llm.listModels(provider: string): Promise` Discover the models one registered provider currently advertises. diff --git a/packages/llm/llm/README.zh.md b/packages/llm/llm/README.zh.md index 9bd26993bc..ca98e875a9 100644 --- a/packages/llm/llm/README.zh.md +++ b/packages/llm/llm/README.zh.md @@ -12,7 +12,7 @@ - `ctx.llm.registerAdapter(providers: string[], adapter: LlmAdapter): AdapterRegistrationHandle` 为给定提供方路由注册一个适配器实例。注册要么全部成功,要么全部不生效,并且会随调用 fiber 一起 dispose(资源释放)。返回的释放器还携带 `replace(providers)`:候选路由集合会在任何东西变动之前完整校验,因此与另一适配器冲突时,当前路由保持注册且继续服务,而替换本身是一个同步区段,不存在可观察的空档。`replace([])` 合法——一个持有零条路由的注册——这与空的初始注册不同。 - `ctx.llm.listProviders(): LlmProviderInfo[]` 按注册顺序描述已注册提供方路由。 -- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void` 声明适配器插件可通过配置激活的提供方路由——无论已注册还是休眠——每个条目指明其所属 settings namespace,以及 profile 在该分节内的路径。要么全部成功,要么全部不生效(`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`),并随调用 fiber dispose。 +- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle` 声明适配器插件可通过配置激活的提供方路由——无论已注册还是休眠——每个条目指明其所属 settings namespace,以及 profile 在该分节内的路径。要么全部成功,要么全部不生效(`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`),并随调用 fiber dispose。该句柄还带 `replace(entries)`:候选集合会先被整体校验,因此其中若有条目已被另一个注册声明,当前集合原封不动;此处允许传空数组。声明集合随配置变化的插件必须使用 `replace`,而不是先 dispose 再重新注册——后者会在新集合被拒时让目录整个落空。 - `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` 按声明顺序列出已声明的目录;配置界面将其与 `listProviders()` 合并,为每个条目标注存活或休眠。 - `ctx.llm.providerRetryPolicy(provider: string): ResolvedRetryPolicy` 返回注册时捕获的提供方重试策略,并解析 normal 默认值。 - `ctx.llm.listModels(provider: string): Promise` 发现某个已注册提供方当前公布的模型。 diff --git a/packages/llm/llm/src/index.ts b/packages/llm/llm/src/index.ts index 8b3c839787..4c1e8e94d5 100644 --- a/packages/llm/llm/src/index.ts +++ b/packages/llm/llm/src/index.ts @@ -225,6 +225,27 @@ export interface AdapterRegistrationHandle { replace(providers: string[]): void } +/** + * A live configurable-provider registration, disposable and atomically + * replaceable — the directory counterpart of {@link AdapterRegistrationHandle}. + */ +export interface DirectoryRegistrationHandle { + /** Withdraw every entry this registration currently holds. */ + (): void + /** + * Replace this registration's entries with `entries`. The candidate set is + * validated in full first — an entry another registration already declares, + * a duplicate within the set, or invalid metadata throws and leaves the + * current entries untouched — and the swap is one synchronous section, so no + * reader observes a gap. An empty array is legal here, unlike an empty + * initial registration. + * + * Throws `LlmError` with code `REGISTRATION_DISPOSED` once the registration + * has been disposed. + */ + replace(entries: readonly LlmConfigurableProvider[]): void +} + /** * The abstract `llm` service: an adapter registry plus a streaming model-call * surface, interceptable via the `llm/stream` waterfall. @@ -370,34 +391,61 @@ export class LlmService extends Service { * entry, or a provider already declared by any registration throws * `LlmError` without registering the rest. Disposed with the fiber. * @param entries - every configurable provider this plugin owns. - * @returns the disposer that withdraws all of them. + * @returns a handle that withdraws all of them, and can atomically replace them. */ - registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): () => void { - const dispose = this.ctx.effect(function* (this: LlmService) { - if (entries.length === 0) { - throw new LlmError('a configurable-provider registration must declare at least one provider', 'INVALID_DIRECTORY') - } + registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle { + let held: LlmConfigurableProvider[] = [] + let disposed = false + /** + * Validate a candidate set in full against everything this registration + * does not already hold, then publish it. Nothing is written until the + * whole set passes, so a refused candidate leaves the current entries in + * place — the property that makes `replace` a swap rather than a + * delete-then-add that can strand the directory empty. + */ + const commit = (candidates: readonly LlmConfigurableProvider[]): void => { const detached: LlmConfigurableProvider[] = [] - for (const entry of entries) { + const own = new Set(held.map(entry => entry.provider)) + for (const entry of candidates) { if (entry.provider.length === 0 || entry.displayName.length === 0 || entry.settingsNs.length === 0) { throw new LlmError('configurable providers need a non-empty provider, displayName, and settingsNs', 'INVALID_DIRECTORY') } if (entry.settingsPath.some(segment => segment.length === 0)) { throw new LlmError(`configurable provider "${entry.provider}" has an empty settingsPath segment`, 'INVALID_DIRECTORY') } - if (this.directory.has(entry.provider) || detached.some(seen => seen.provider === entry.provider)) { + if ((this.directory.has(entry.provider) && !own.has(entry.provider)) + || detached.some(seen => seen.provider === entry.provider)) { throw new LlmError(`configurable provider "${entry.provider}" is already declared`, 'DUPLICATE_DIRECTORY') } detached.push({ ...entry, settingsPath: [...entry.settingsPath] }) } + for (const entry of held) this.directory.delete(entry.provider) for (const entry of detached) this.directory.set(entry.provider, entry) + held = detached this.emitAdaptersUpdated() + } + + const dispose = this.ctx.effect(function* (this: LlmService) { + if (entries.length === 0) { + throw new LlmError('a configurable-provider registration must declare at least one provider', 'INVALID_DIRECTORY') + } + commit(entries) yield () => { - for (const entry of detached) this.directory.delete(entry.provider) + disposed = true + for (const entry of held) this.directory.delete(entry.provider) + held = [] this.emitAdaptersUpdated() } }.bind(this), 'llm.registerConfigurableProviders()') - return () => void dispose() + + const handle = ((): void => void dispose()) as DirectoryRegistrationHandle + handle.replace = (next: readonly LlmConfigurableProvider[]): void => { + if (disposed) { + throw new LlmError('this configurable-provider registration was disposed', 'REGISTRATION_DISPOSED') + } + commit(next) + } + return handle } /** diff --git a/packages/llm/llm/tests/topology.spec.ts b/packages/llm/llm/tests/topology.spec.ts index f07b33af7d..3e54db480f 100644 --- a/packages/llm/llm/tests/topology.spec.ts +++ b/packages/llm/llm/tests/topology.spec.ts @@ -170,6 +170,32 @@ describe('configurable-provider directory', () => { expect(ctx.llm.listConfigurableProviders()).toEqual([]) }) + it('replaces its entries atomically, keeping the old set when a candidate collides', async () => { + const ctx = await setup() + const handle = ctx.llm.registerConfigurableProviders([entry(), entry({ provider: 'second' })]) + ctx.llm.registerConfigurableProviders([entry({ provider: 'owned-elsewhere' })]) + + // A candidate another registration already declares refuses the whole swap. + expect(() =>{ handle.replace([entry({ provider: 'owned-elsewhere' })]); }).toThrow(/already declared/) + expect(ctx.llm.listConfigurableProviders().map(view => view.provider).sort()) + .toEqual(['owned-elsewhere', 'second', entry().provider].sort()) + + // Its own entries are not "already declared" against itself, so a swap that + // keeps one and drops another lands whole. + handle.replace([entry({ displayName: 'Renamed' })]) + expect(ctx.llm.listConfigurableProviders().map(view => view.provider).sort()) + .toEqual(['owned-elsewhere', entry().provider].sort()) + expect(ctx.llm.listConfigurableProviders().find(view => view.provider === entry().provider)?.displayName) + .toBe('Renamed') + + // An empty replace is legal, unlike an empty initial registration. + handle.replace([]) + expect(ctx.llm.listConfigurableProviders().map(view => view.provider)).toEqual(['owned-elsewhere']) + + handle() + expect(() =>{ handle.replace([entry()]); }).toThrow(/was disposed/) + }) + it('rejects duplicates within one registration and across registrations', async () => { const ctx = await setup() expect(() => ctx.llm.registerConfigurableProviders([entry(), entry()])).toThrow(/already declared/) diff --git a/scripts/gen-cordis-catalog.ts b/scripts/gen-cordis-catalog.ts index 4dd381af69..30e822910e 100644 --- a/scripts/gen-cordis-catalog.ts +++ b/scripts/gen-cordis-catalog.ts @@ -34,6 +34,7 @@ export const LINK_MAP: Readonly> = { HookContext: 'core.md', SettleReason: 'core.md', AdapterRegistrationHandle: 'core.md', + DirectoryRegistrationHandle: 'core.md', LlmCallConfig: 'core.md', LlmModelContext: 'core.md', LlmModelReasoningInfo: 'core.md',