refactor(llm): resolve model metadata together

This commit is contained in:
Yichen Jiang
2026-07-26 13:07:27 +08:00
parent baea5018e5
commit 73e7e27799
55 changed files with 607 additions and 460 deletions

View File

@@ -28,9 +28,9 @@ The package root exposes the Cordis plugin contract and `DeepSeekAdapter`; wire
The plugin registers the single provider route `deepseek`. A request selects it with `provider: deepseek`; its `model` is passed through as the wire `model` string, so changing DeepSeek models does not require lifecycle-time registration. Omitting `models` advertises `deepseek-v4-flash` and `deepseek-v4-pro`, each with a 128,000-token context window; an explicit list replaces those defaults, while `models: []` advertises none. Catalog entries are exposed through `ctx.llm.listModels('deepseek')` for UI selectors and deployment introspection, but remain advisory: unlisted model ids still pass through unchanged. An omitted entry name defaults to its id.
`contextWindow` is optional per configured model and is not exposed through the advisory catalog. `ctx.llm.resolveModelContext('deepseek', model)` returns an exact model value first, then `defaultContextWindow` for an entry without capacity or an unlisted pass-through id. When neither value exists it returns `undefined` without invalidating routing. Pressure-sensitive plugins therefore get deployment-owned capacity without treating the model selector as authoritative. Registering another adapter for `deepseek` throws `LlmError('DUPLICATE_ADAPTER')`.
`contextWindow` is optional per configured model and is not exposed through the advisory catalog. `ctx.llm.resolveModelInfo('deepseek', model).context` returns an exact model value first, then `defaultContextWindow` for an entry without capacity or an unlisted pass-through id. When neither value exists, `context` is absent without invalidating routing. Pressure-sensitive plugins therefore get deployment-owned capacity without treating the model selector as authoritative. Registering another adapter for `deepseek` throws `LlmError('DUPLICATE_ADAPTER')`.
`ctx.llm.resolveModelReasoning('deepseek', model)` returns the ordered `high` and `max` efforts for every pass-through model while thinking is enabled. `reasoningEffort` selects the deployment default and falls back to `high` when omitted. `agent/request` can replace it on each conversation step; the resolved value is logged in `request/header` and serialized as the official top-level `reasoning_effort` field. An unsupported value fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O.
The same exact-model result exposes ordered `high` and `max` efforts under `reasoning` for every pass-through model while thinking is enabled. `reasoningEffort` selects the deployment default and falls back to `high` when omitted. `agent/request` can replace it on each conversation step; the resolved value is logged in `request/header` and serialized as the official top-level `reasoning_effort` field. An unsupported value fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O.
`thinking: disabled` removes the reasoning capability and omits `reasoning_effort`; combining it with a configured default fails plugin loading, and a per-request effort fails as unsupported. A request with `GenerateOptions.purpose: 'session-title'` also forces thinking disabled and omits the already-resolved effort, reserving its bounded output for visible title text without changing conversation or compaction defaults.

View File

@@ -8,10 +8,9 @@
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId } from '@deepseek-ai/dsh-llm'
import type {
GenerateOptions,
LlmModelContext,
LlmModelInfo,
LlmModelReasoningInfo,
LlmProviderInfo,
LlmResolvedModelInfo,
StreamChunk,
} from '@deepseek-ai/dsh-llm'
import { idleWatchdog, MAX_TIMER_DELAY_MS, timeoutOf } from '@deepseek-ai/dsh-timeout'
@@ -59,6 +58,15 @@ const REASONING_EFFORTS = [
{ id: MAX_REASONING_EFFORT, name: 'Max' },
] as const
function modelInfo(provider: string, model: DeepSeekCatalogModel): LlmModelInfo {
return {
provider,
id: model.id,
name: model.name ?? model.id,
...model.description === undefined ? {} : { description: model.description },
}
}
function providerRetryAfterMs(value: string | null): number | undefined {
if (value === null) return undefined
if (/^\d+$/.test(value)) {
@@ -127,34 +135,32 @@ export class DeepSeekAdapter extends LlmAdapter {
}
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
return Promise.resolve((this.options.models ?? []).map(model => ({
provider,
id: model.id,
name: model.name ?? model.id,
...model.description === undefined ? {} : { description: model.description },
})))
return Promise.resolve((this.options.models ?? []).map(model => modelInfo(provider, model)))
}
override resolveModelContext(
_provider: string,
override resolveModel(
provider: string,
model: string,
): Promise<LlmModelContext | undefined> {
const contextWindow = this.options.models?.find(entry => entry.id === model)?.contextWindow
?? this.options.defaultContextWindow
return Promise.resolve(contextWindow === undefined ? undefined : { contextWindow })
}
override resolveModelReasoning(
_provider: string,
_model: string,
_signal?: AbortSignal,
): Promise<LlmModelReasoningInfo | undefined> {
if (this.options.defaults?.thinking === 'disabled') return Promise.resolve(undefined)
): Promise<LlmResolvedModelInfo> {
const configured = this.options.models?.find(entry => entry.id === model)
const contextWindow = configured?.contextWindow
?? this.options.defaultContextWindow
return Promise.resolve({
efforts: REASONING_EFFORTS,
defaultEffort: this.options.defaults?.reasoningEffort === 'max'
? MAX_REASONING_EFFORT
: HIGH_REASONING_EFFORT,
...configured === undefined
? { provider, id: model, name: model }
: modelInfo(provider, configured),
...contextWindow === undefined ? {} : { context: { contextWindow } },
...this.options.defaults?.thinking === 'disabled'
? {}
: {
reasoning: {
efforts: REASONING_EFFORTS,
defaultEffort: this.options.defaults?.reasoningEffort === 'max'
? MAX_REASONING_EFFORT
: HIGH_REASONING_EFFORT,
},
},
})
}

View File

@@ -213,8 +213,8 @@ describe('DeepSeekAdapter against a mock server', () => {
thinking: { type: 'disabled' },
})
expect(server.requests[0]).not.toHaveProperty('reasoning_effort')
await expect(ctx.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash'))
.resolves.toBeUndefined()
await expect(ctx.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
.resolves.not.toHaveProperty('reasoning')
})
it('rejects a per-request effort before I/O when thinking is disabled', async () => {
@@ -573,15 +573,19 @@ describe('plugin registration and config', () => {
{ provider: 'deepseek', id: 'deepseek-v4-flash', name: 'deepseek-v4-flash' },
{ provider: 'deepseek', id: 'deepseek-v4-pro', name: 'deepseek-v4-pro' },
])
await expect(ctx.llm.resolveModelContext('deepseek', 'deepseek-v4-flash'))
.resolves.toEqual({ contextWindow: 128_000 })
await expect(ctx.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash'))
.resolves.toEqual({
efforts: [
{ id: ReasoningEffortId('high'), name: 'High' },
{ id: ReasoningEffortId('max'), name: 'Max' },
],
defaultEffort: ReasoningEffortId('high'),
await expect(ctx.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
.resolves.toMatchObject({
provider: 'deepseek',
id: 'deepseek-v4-flash',
name: 'deepseek-v4-flash',
context: { contextWindow: 128_000 },
reasoning: {
efforts: [
{ id: ReasoningEffortId('high'), name: 'High' },
{ id: ReasoningEffortId('max'), name: 'Max' },
],
defaultEffort: ReasoningEffortId('high'),
},
})
})
@@ -635,10 +639,15 @@ describe('plugin registration and config', () => {
{ provider: 'deepseek', id: 'private-fast', name: 'private-fast' },
{ provider: 'deepseek', id: 'private-reasoner', name: 'Private Reasoner', description: 'Higher reasoning budget' },
])
await expect(ctx.llm.resolveModelContext('deepseek', 'private-fast'))
.resolves.toEqual({ contextWindow: 32_000 })
await expect(ctx.llm.resolveModelContext('deepseek', 'arbitrary-unlisted'))
.resolves.toBeUndefined()
await expect(ctx.llm.resolveModelInfo('deepseek', 'private-fast'))
.resolves.toMatchObject({ context: { contextWindow: 32_000 } })
await expect(ctx.llm.resolveModelInfo('deepseek', 'private-reasoner'))
.resolves.toMatchObject({
name: 'Private Reasoner',
description: 'Higher reasoning budget',
})
await expect(ctx.llm.resolveModelInfo('deepseek', 'arbitrary-unlisted'))
.resolves.not.toHaveProperty('context')
})
it('uses exact model capacity before the adapter-wide default', async () => {
@@ -654,12 +663,12 @@ describe('plugin registration and config', () => {
],
})
await expect(ctx.llm.resolveModelContext('deepseek', 'inherits-default'))
.resolves.toEqual({ contextWindow: 256_000 })
await expect(ctx.llm.resolveModelContext('deepseek', 'exact-override'))
.resolves.toEqual({ contextWindow: 64_000 })
await expect(ctx.llm.resolveModelContext('deepseek', 'unlisted-pass-through'))
.resolves.toEqual({ contextWindow: 256_000 })
await expect(ctx.llm.resolveModelInfo('deepseek', 'inherits-default'))
.resolves.toMatchObject({ context: { contextWindow: 256_000 } })
await expect(ctx.llm.resolveModelInfo('deepseek', 'exact-override'))
.resolves.toMatchObject({ context: { contextWindow: 64_000 } })
await expect(ctx.llm.resolveModelInfo('deepseek', 'unlisted-pass-through'))
.resolves.toMatchObject({ context: { contextWindow: 256_000 } })
})
it('allows an explicit empty model catalog', async () => {

View File

@@ -28,9 +28,9 @@ Configure credentials and deployment-specific transport settings per provider. O
Each provider name must exist in pi-ai's installed catalog and may appear only once in this plugin instance. Registration with `ctx.llm` is atomic: a collision with any provider route already owned by another adapter fails plugin loading without registering the remaining routes. Model ids are not lifecycle config; an unknown model fails before any provider request with `LlmError('UNKNOWN_MODEL')`.
The adapter exposes each configured provider's installed pi-ai models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata derived from `getModels(provider)`; request-time resolution still performs the authoritative catalog lookup, so discovery does not create a second model registry. `ctx.llm.resolveModelContext(provider, model)` performs the same exact descriptor lookup and returns its context window, keeping capacity metadata on the route-owning adapter rather than a consuming plugin.
The adapter exposes each configured provider's installed pi-ai models through `ctx.llm.listModels(provider)`. This is provider-neutral selector metadata derived from `getModels(provider)`; request-time resolution still performs the authoritative catalog lookup, so discovery does not create a second model registry. `ctx.llm.resolveModelInfo(provider, model)` performs that exact descriptor lookup once and returns its identity, context window, and selectable thinking levels, keeping authoritative metadata on the route-owning adapter rather than its consumers.
`ctx.llm.resolveModelReasoning(provider, model)` uses pi-ai's `getSupportedThinkingLevels(model)` and returns that model's ordered levels after filtering the separate `off` control. The Harness exposes the canonical pi-ai level as an opaque ID; provider/model wire spellings remain inside pi-ai's `thinkingLevelMap`. A non-reasoning model returns `undefined`. The profile `reasoning` value is the deployment default when configured; omitting it preserves the provider default. Per-request `GenerateOptions.reasoningEffort` takes precedence, and any explicit value absent from the exact model capability fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O instead of being clamped.
The `reasoning.efforts` list is pi-ai's ordered `getSupportedThinkingLevels(model)` result without filtering or normalization, including `off` and the model-specific availability of `xhigh` or `max`. The Harness exposes each canonical pi-ai level as an opaque ID; provider/model wire spellings remain inside pi-ai's `thinkingLevelMap`. A non-reasoning model therefore exposes pi-ai's `off` choice. The profile `reasoning` value, including `off`, is the deployment default when configured; omitting it preserves the provider default. Per-request `GenerateOptions.reasoningEffort` takes precedence, and any explicit value absent from the exact model capability fails with `UNSUPPORTED_REASONING_EFFORT` before network I/O instead of being clamped. pi-ai's common stream options represent `off` by omitting `reasoning`.
Supported profile fields are `provider`, `apiKey`, `baseURL`, `headers`, `reasoning`, `thinkingBudgets`, `cacheRetention`, `transport`, `timeoutMs`, `websocketConnectTimeoutMs`, and `streamIdleTimeoutMs`. The stream-idle interval is a positive finite Node timer delay, defaults to five minutes, and covers only an outstanding provider read, not consumer think time. Harness app attribution wins a conflicting configured header name.
@@ -49,7 +49,7 @@ If a listener rewrites assembled assistant content, the loop drops replay state
- pi-ai tool-call arguments are parsed objects; the harness stores raw JSON strings. The adapter parses input and re-stringifies output.
- pi-ai reports failures as in-stream error events; these map to `finish {kind:'error'|'aborted', failure}` chunks. Provider-specific error text distinguishes terminal `QUOTA` from transient `RATE_LIMIT`, while text and usage signals evaluated against the resolved model's context window normalize overflow to `CONTEXT_WINDOW_EXCEEDED`. A terminal `stop` whose message carries no content blocks maps to a `finish {kind:'error'}` with code `EMPTY_RESPONSE` (retried by default policy) instead of a successful empty message.
- pi-ai folds reasoning tokens into output usage; there is no separate reasoning count to map.
- pi-ai may internally support an `off` thinking level, but the Harness reasoning-effort capability deliberately excludes mode changes.
- pi-ai's `off` thinking level crosses the Harness capability seam unchanged and becomes an omitted pi-ai common `reasoning` option at dispatch.
- `GenerateOptions.stop` is rejected with `UNSUPPORTED_OPTION` because pi-ai's common streaming surface cannot guarantee it across providers.
## App attribution

View File

@@ -11,6 +11,7 @@ import { getSupportedThinkingLevels } from '@earendil-works/pi-ai'
import type {
Api,
Model,
ModelThinkingLevel,
SimpleStreamOptions,
ThinkingLevel,
} from '@earendil-works/pi-ai'
@@ -22,9 +23,8 @@ import {
} from '@deepseek-ai/dsh-llm'
import type {
GenerateOptions,
LlmModelContext,
LlmModelInfo,
LlmModelReasoningInfo,
LlmResolvedModelInfo,
ReasoningEffortId as ReasoningEffortIdType,
StreamChunk,
} from '@deepseek-ai/dsh-llm'
@@ -44,7 +44,7 @@ export interface PiAiAdapterOptions {
* Resolve a catalog model dynamically and apply only the configured endpoint
* override, preserving the catalog's API/capability/compatibility metadata.
*/
function resolveModel(profile: PiAiProviderProfile, modelId: string): Model<Api> {
function resolvePiModel(profile: PiAiProviderProfile, modelId: string): Model<Api> {
const model = getBuiltinModels(profile.provider as BuiltinProvider).find(candidate => candidate.id === modelId) as Model<Api> | undefined
if (model === undefined) {
throw new LlmError(`pi-ai provider "${profile.provider}" has no catalog model "${modelId}"`, 'UNKNOWN_MODEL')
@@ -55,11 +55,12 @@ function resolveModel(profile: PiAiProviderProfile, modelId: string): Model<Api>
/** Copy profile stream knobs into pi-ai's common option vocabulary. */
function profileOptions(
profile: PiAiProviderProfile,
reasoning: ThinkingLevel | undefined,
reasoning: ModelThinkingLevel | undefined,
): SimpleStreamOptions {
const enabledReasoning: ThinkingLevel | undefined = reasoning === 'off' ? undefined : reasoning
return {
...profile.apiKey === undefined ? {} : { apiKey: profile.apiKey },
...reasoning === undefined ? {} : { reasoning },
...enabledReasoning === undefined ? {} : { reasoning: enabledReasoning },
...profile.thinkingBudgets === undefined ? {} : { thinkingBudgets: profile.thinkingBudgets },
...profile.cacheRetention === undefined ? {} : { cacheRetention: profile.cacheRetention },
...profile.transport === undefined ? {} : { transport: profile.transport },
@@ -70,19 +71,14 @@ function profileOptions(
}
}
/** Selectable pi-ai levels exclude the separate on/off control. */
function supportedReasoningLevels(model: Model<Api>): ThinkingLevel[] {
return getSupportedThinkingLevels(model).filter((level): level is ThinkingLevel => level !== 'off')
}
/** Validate an explicit Harness/profile effort without invoking pi-ai's clamp. */
function resolveReasoningLevel(
model: Model<Api>,
effort: ReasoningEffortIdType | ThinkingLevel | undefined,
): ThinkingLevel | undefined {
effort: ReasoningEffortIdType | ModelThinkingLevel | undefined,
): ModelThinkingLevel | undefined {
if (effort === undefined) return undefined
const supported = supportedReasoningLevels(model)
if (supported.some(level => level === effort)) return effort as ThinkingLevel
const supported = getSupportedThinkingLevels(model)
if (supported.some(level => level === effort)) return effort as ModelThinkingLevel
throw new LlmError(
`pi-ai provider "${model.provider}" model "${model.id}" does not support reasoning effort "${effort}"`,
'UNSUPPORTED_REASONING_EFFORT',
@@ -123,27 +119,11 @@ export class PiAiAdapter extends LlmAdapter {
})))
}
override resolveModelContext(
provider: string,
model: string,
): Promise<LlmModelContext | undefined> {
const profile = this.profiles.get(provider)
if (profile === undefined) {
return Promise.reject(new LlmError(
`pi-ai adapter does not own provider "${provider}"`,
'NO_ADAPTER',
))
}
return Promise.resolve().then(() => ({
contextWindow: resolveModel(profile, model).contextWindow,
}))
}
override resolveModelReasoning(
override resolveModel(
provider: string,
model: string,
_signal?: AbortSignal,
): Promise<LlmModelReasoningInfo | undefined> {
): Promise<LlmResolvedModelInfo> {
const profile = this.profiles.get(provider)
if (profile === undefined) {
return Promise.reject(new LlmError(
@@ -152,21 +132,23 @@ export class PiAiAdapter extends LlmAdapter {
))
}
return Promise.resolve().then(() => {
const resolvedModel = resolveModel(profile, model)
const levels = supportedReasoningLevels(resolvedModel)
if (levels.length === 0) {
resolveReasoningLevel(resolvedModel, profile.reasoning)
return undefined
}
const resolvedModel = resolvePiModel(profile, model)
const levels = getSupportedThinkingLevels(resolvedModel)
const defaultLevel = resolveReasoningLevel(resolvedModel, profile.reasoning)
return {
efforts: levels.map(level => ({
id: ReasoningEffortId(level),
name: `${level.charAt(0).toUpperCase()}${level.slice(1)}`,
})),
...defaultLevel === undefined
? {}
: { defaultEffort: ReasoningEffortId(defaultLevel) },
provider,
id: model,
name: resolvedModel.name,
context: { contextWindow: resolvedModel.contextWindow },
reasoning: {
efforts: levels.map(level => ({
id: ReasoningEffortId(level),
name: `${level.charAt(0).toUpperCase()}${level.slice(1)}`,
})),
...defaultLevel === undefined
? {}
: { defaultEffort: ReasoningEffortId(defaultLevel) },
},
}
})
}
@@ -179,7 +161,7 @@ export class PiAiAdapter extends LlmAdapter {
if (profile === undefined) {
throw new LlmError(`pi-ai adapter does not own provider "${options.provider}"`, 'NO_ADAPTER')
}
const model = resolveModel(profile, options.model)
const model = resolvePiModel(profile, options.model)
const reasoning = resolveReasoningLevel(
model,
options.reasoningEffort ?? profile.reasoning,

View File

@@ -5,7 +5,7 @@
*/
import { getBuiltinProviders } from '@earendil-works/pi-ai/providers/all'
import type { CacheRetention, ThinkingBudgets, ThinkingLevel, Transport } from '@earendil-works/pi-ai'
import type { CacheRetention, ModelThinkingLevel, ThinkingBudgets, Transport } from '@earendil-works/pi-ai'
import z from 'schemastery'
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
@@ -23,7 +23,7 @@ export interface PiAiProviderProfile {
/** Provider request headers; Harness attribution wins reserved names. */
headers?: Record<string, string>
/** Provider-neutral pi-ai reasoning level. */
reasoning?: ThinkingLevel
reasoning?: ModelThinkingLevel
/** Token budgets used by reasoning providers that support them. */
thinkingBudgets?: ThinkingBudgets
/** Prompt-cache retention preference. */
@@ -62,7 +62,7 @@ const profile = z.object({
apiKey: z.string(),
baseURL: z.string(),
headers: z.dict(z.string()),
reasoning: z.union(['minimal', 'low', 'medium', 'high', 'xhigh', 'max']),
reasoning: z.union(['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max']),
thinkingBudgets,
cacheRetention: z.union(['none', 'short', 'long']),
transport: z.union(['sse', 'websocket', 'websocket-cached', 'auto']),

View File

@@ -9,7 +9,7 @@ import { assemble, type AssembledResult } from './assemble.ts'
/**
* Real-API e2e for the pi-ai-backed adapter: V4 Flash + V4 Pro with provider
* defaults and representative high/max reasoning. Mirrors the native
* defaults and representative off/high/max reasoning. Mirrors the native
* adapter's StreamChunk contract and exercises a replayed tool follow-up.
* Key-gated.
*/
@@ -74,6 +74,19 @@ describe.skipIf(!process.env.DEEPSEEK_API_KEY)('llm-pi-ai e2e (real API)', () =>
expect(textOf(result).toLowerCase()).toContain('pong')
})
it('flash + reasoning off: plain text without reasoning blocks', async () => {
const ctx = await harness(FLASH)
const result = await assemble(ctx,{
model: FLASH,
reasoningEffort: ReasoningEffortId('off'),
messages: ask('Reply with exactly the word: pong'),
maxTokens: 50,
})
expect(result.finish.kind).toBe('stop')
expect(result.message.content.some(block => block.type === 'reasoning')).toBe(false)
expect(textOf(result).toLowerCase()).toContain('pong')
})
it.each([FLASH, PRO])('%s + reasoning high: reasoning blocks present', async (model) => {
const ctx = await harness(model)
const result = await assemble(ctx,{

View File

@@ -149,7 +149,7 @@ describe('PiAiAdapter provider routing', () => {
})
it('uses a dynamic request effort and rejects unsupported efforts before network I/O', async () => {
const server = await mockServer([{ events: textEvents }])
const server = await mockServer([{ events: textEvents }, { events: textEvents }])
const ctx = await harness(server.url, { reasoning: 'max' })
await assemble(ctx, {
@@ -159,12 +159,20 @@ describe('PiAiAdapter provider routing', () => {
})
expect(server.requests[0]).toMatchObject({ reasoning_effort: 'high' })
await assemble(ctx, {
model: 'deepseek-v4-flash',
reasoningEffort: ReasoningEffortId('off'),
messages: [],
})
expect(server.requests[1]).toMatchObject({ thinking: { type: 'disabled' } })
expect(server.requests[1]).not.toHaveProperty('reasoning_effort')
await expect(assemble(ctx, {
model: 'deepseek-v4-flash',
reasoningEffort: ReasoningEffortId('xhigh'),
messages: [],
})).rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' })
expect(server.requests).toHaveLength(1)
expect(server.requests).toHaveLength(2)
})
it('preserves omitted profile options when constructing the adapter directly', async () => {
@@ -341,27 +349,30 @@ describe('provider profile lifecycle', () => {
provider: 'openai', id: 'gpt-4.1', name: 'GPT-4.1',
})
expect(models.every(model => model.provider === 'openai')).toBe(true)
const context = await ctx.llm.resolveModelContext('openai', 'gpt-4.1')
expect(context).toBeDefined()
expect(typeof context?.contextWindow).toBe('number')
const info = await ctx.llm.resolveModelInfo('openai', 'gpt-4.1')
expect(typeof info.context?.contextWindow).toBe('number')
})
it('exposes model-specific reasoning levels without off or an invented provider default', async () => {
it('exposes pi-ai model thinking levels verbatim without inventing a provider default', async () => {
const ctx = new Context()
await ctx.plugin(LlmService)
await ctx.plugin(LlmPiAi, {
providers: [{ provider: 'deepseek' }, { provider: 'openai' }],
})
await expect(ctx.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash'))
.resolves.toEqual({
efforts: [
{ id: ReasoningEffortId('high'), name: 'High' },
{ id: ReasoningEffortId('max'), name: 'Max' },
],
await expect(ctx.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
.resolves.toMatchObject({
reasoning: {
efforts: [
{ id: ReasoningEffortId('off'), name: 'Off' },
{ id: ReasoningEffortId('high'), name: 'High' },
{ id: ReasoningEffortId('max'), name: 'Max' },
],
},
})
const extended = await ctx.llm.resolveModelReasoning('openai', 'gpt-5.6-sol')
expect(extended?.efforts.map(effort => effort.id)).toEqual([
const extended = await ctx.llm.resolveModelInfo('openai', 'gpt-5.6-sol')
expect(extended.reasoning?.efforts.map(effort => effort.id)).toEqual([
ReasoningEffortId('off'),
ReasoningEffortId('minimal'),
ReasoningEffortId('low'),
ReasoningEffortId('medium'),
@@ -369,8 +380,12 @@ describe('provider profile lifecycle', () => {
ReasoningEffortId('xhigh'),
ReasoningEffortId('max'),
])
await expect(ctx.llm.resolveModelReasoning('openai', 'gpt-4.1'))
.resolves.toBeUndefined()
await expect(ctx.llm.resolveModelInfo('openai', 'gpt-4.1'))
.resolves.toMatchObject({
reasoning: {
efforts: [{ id: ReasoningEffortId('off'), name: 'Off' }],
},
})
})
it('uses a supported profile reasoning value as the model default and rejects an unsupported one', async () => {
@@ -379,16 +394,24 @@ describe('provider profile lifecycle', () => {
await supported.plugin(LlmPiAi, {
providers: [{ provider: 'deepseek', reasoning: 'max' }],
})
await expect(supported.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash'))
.resolves.toMatchObject({ defaultEffort: ReasoningEffortId('max') })
await expect(supported.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
.resolves.toMatchObject({ reasoning: { defaultEffort: ReasoningEffortId('max') } })
const unsupported = new Context()
await unsupported.plugin(LlmService)
await unsupported.plugin(LlmPiAi, {
providers: [{ provider: 'deepseek', reasoning: 'medium' }],
})
await expect(unsupported.llm.resolveModelReasoning('deepseek', 'deepseek-v4-flash'))
await expect(unsupported.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
.rejects.toMatchObject({ code: 'UNSUPPORTED_REASONING_EFFORT' })
const disabled = new Context()
await disabled.plugin(LlmService)
await disabled.plugin(LlmPiAi, {
providers: [{ provider: 'deepseek', reasoning: 'off' }],
})
await expect(disabled.llm.resolveModelInfo('deepseek', 'deepseek-v4-flash'))
.resolves.toMatchObject({ reasoning: { defaultEffort: ReasoningEffortId('off') } })
})
it('accepts absent credentials for pi-ai ambient authentication', async () => {
@@ -440,11 +463,9 @@ describe('provider profile lifecycle', () => {
it('constructs the adapter directly and rejects routes it does not own', async () => {
const adapter = new PiAiAdapter({ profiles: [{ provider: 'openai' }] })
await expect(adapter.listModels('anthropic')).rejects.toMatchObject({ code: 'NO_ADAPTER' })
await expect(adapter.resolveModelContext('anthropic', 'claude-sonnet-4'))
await expect(adapter.resolveModel('anthropic', 'claude-sonnet-4'))
.rejects.toMatchObject({ code: 'NO_ADAPTER' })
await expect(adapter.resolveModelReasoning('anthropic', 'claude-sonnet-4'))
.rejects.toMatchObject({ code: 'NO_ADAPTER' })
await expect(adapter.resolveModelContext('openai', 'not-a-catalog-model'))
await expect(adapter.resolveModel('openai', 'not-a-catalog-model'))
.rejects.toMatchObject({ code: 'UNKNOWN_MODEL' })
await expect((async () => {
for await (const _chunk of adapter.stream({ provider: 'anthropic', model: 'claude-sonnet-4', messages: [] })) { /* drain */ }

View File

@@ -11,8 +11,7 @@ An adapter registry plus a single streaming call surface, interceptable via a wa
- `ctx.llm.registerAdapter(providers: string[], adapter: LlmAdapter): () => void` Register one adapter instance for the given provider routes. Registration is all-or-nothing, and is disposed with the calling fiber.
- `ctx.llm.listProviders(): LlmProviderInfo[]` Describe registered provider routes in registration order.
- `ctx.llm.listModels(provider: string): Promise<LlmModelInfo[]>` Discover the models one registered provider currently advertises.
- `ctx.llm.resolveModelContext(provider: string, model: string): Promise<LlmModelContext | undefined>` Resolve authoritative context capacity for one exact route from its owning adapter.
- `ctx.llm.resolveModelReasoning(provider: string, model: string, signal?: AbortSignal): Promise<LlmModelReasoningInfo | undefined>` Resolve ordered adapter-owned reasoning efforts and an optional deployment default for one exact route, with optional cancellation for asynchronous adapters.
- `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<LlmResolvedModelInfo>` Resolve validated exact-model identity plus available context and reasoning metadata from the owning adapter, with optional cancellation for asynchronous adapters.
- `ctx.llm.resolveCallConfig(config: LlmCallConfig, signal?: AbortSignal): Promise<LlmCallConfig>` Validate an explicit effort and materialize an adapter-configured default without clamping.
- `ctx.llm.prepareCall(config: LlmCallConfig, signal?: AbortSignal): Promise<PreparedLlmCall>` Resolve a config and capture its current adapter registration as one cancellable, one-shot call.
- `ctx.llm.stream(options: GenerateOptions): AsyncIterable<StreamChunk>` Stream one model call as raw chunks (token-level deltas). Consumers assemble the chunks into blocks/messages with `BlockAssembler`.
@@ -21,9 +20,9 @@ An adapter registry plus a single streaming call surface, interceptable via a wa
Provider and model metadata is a discovery surface, not a routing whitelist. `registerAdapter()` still owns provider exclusivity, while an adapter may accept model ids absent from `listModels()`; consumers must not reject a request because its model is unlisted. Returned metadata is detached and invalid or duplicate adapter entries fail with `INVALID_ADAPTER` or `INVALID_CATALOG`.
Context capacity is a separate correctness query, not a catalog decoration or global LLM setting. `resolveModelContext()` asks the adapter that owns the exact provider/model route; an adapter can describe an unlisted dynamic model, and `undefined` means only that capacity is unavailable. Invalid returned capacity fails with `INVALID_MODEL_CONTEXT`.
Exact-model metadata is a separate correctness query, not a catalog decoration or global LLM setting. `resolveModelInfo()` asks the adapter that owns the exact provider/model route once; an adapter can describe an unlisted dynamic model, and absent `context` or `reasoning` fields mean only that those capabilities are unavailable. Invalid identity, context, or reasoning metadata fails with `INVALID_MODEL_INFO`, `INVALID_MODEL_CONTEXT`, or `INVALID_MODEL_REASONING`.
Reasoning effort is also an exact-route capability, but its identifiers are opaque adapter-owned strings rather than a core enum. `resolveModelReasoning()` validates and detaches the ordered display metadata; `undefined` means the model has no selectable effort. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Asynchronous resolvers receive the caller's signal and must settle promptly after cancellation. `prepareCall()` additionally retains the exact adapter registration through header logging and terminal dispatch, so HMR cannot combine one adapter's capability result with another adapter's request; reusing its one-shot handle or changing its call-config fields fails with `INVALID_PREPARED_CALL`. Invalid capability metadata fails with `INVALID_MODEL_REASONING`; an unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O.
Reasoning identifiers are opaque adapter-owned strings rather than a core enum. An adapter publishes its ordered selectable list, including an `off` id when that model's capability API exposes one. `resolveCallConfig()` accepts only an exact advertised identifier, materializes `defaultEffort` when present, and otherwise preserves the provider default. Asynchronous model resolvers receive the caller's signal and must settle promptly after cancellation. `prepareCall()` additionally retains the exact adapter registration through header logging and terminal dispatch, so HMR cannot combine one adapter's capability result with another adapter's request; reusing its one-shot handle or changing its call-config fields fails with `INVALID_PREPARED_CALL`. An unsupported explicit or configured effort fails with `UNSUPPORTED_REASONING_EFFORT` before provider I/O.
### Events
@@ -33,7 +32,7 @@ Reasoning effort is also an exact-route capability, but its identifiers are opaq
### Extension points
- Subclass `LlmAdapter` and call `ctx.llm.registerAdapter(providers, adapter)` to add one or more provider routes. `GenerateOptions.provider` selects the adapter; `GenerateOptions.model` is adapter-owned and may be resolved dynamically. Override `providerInfo()` and asynchronous `listModels()` to expose selector metadata, `resolveModelContext()` when exact capacity is known, and `resolveModelReasoning()` when a model exposes selectable efforts; an asynchronous reasoning resolver must honor its optional cancellation signal. The defaults use the route id as its name, advertise no models, and return neither capacity nor reasoning metadata.
- Subclass `LlmAdapter` and call `ctx.llm.registerAdapter(providers, adapter)` to add one or more provider routes. `GenerateOptions.provider` selects the adapter; `GenerateOptions.model` is adapter-owned and may be resolved dynamically. Override `providerInfo()` and asynchronous `listModels()` to expose selector metadata, then implement `resolveModel()` when exact identity, capacity, or selectable reasoning efforts are available; an asynchronous resolver must honor its optional cancellation signal. The defaults use the route and model ids as names, advertise no models, and return no capacity or reasoning metadata.
- Wrap `llm/stream` via `ctx.on()` waterfall listeners for caching, logging, or routing. A wrapper that retries after emitting a chunk has no durable attempt boundary; shipped agent retry policy therefore uses `agent/request-error` instead.
### Content-block vocabulary (`types.ts`)

View File

@@ -10,9 +10,8 @@ import { Context, Service } from 'cordis'
import type {
GenerateOptions,
LlmFailure,
LlmModelContext,
LlmModelInfo,
LlmModelReasoningInfo,
LlmResolvedModelInfo,
LlmProviderInfo,
Message,
StreamChunk,
@@ -147,34 +146,20 @@ export abstract class LlmAdapter {
}
/**
* Resolve context capacity for one model accepted by this adapter. Absence
* means the adapter does not know the capacity, not that routing is invalid.
* @param _provider - one provider route owned by this adapter.
* @param _model - exact model id passed to {@link GenerateOptions.model}.
* @returns provider-owned context metadata, or `undefined` when unavailable.
* Resolve all metadata available for one exact model. This query is
* independent of the advisory catalog and does not validate request routing.
* @param provider - one provider route owned by this adapter.
* @param model - exact model id passed to {@link GenerateOptions.model}.
* @param _signal - cancellation for this exact-model lookup; asynchronous
* implementations must settle promptly after it aborts.
* @returns provider/model identity plus any context and reasoning metadata.
*/
resolveModelContext(
_provider: string,
_model: string,
): Promise<LlmModelContext | undefined> {
return Promise.resolve(undefined)
}
/**
* Resolve selectable reasoning efforts for one exact model. Absence means
* the model has no selectable reasoning-effort capability.
* @param _provider - one provider route owned by this adapter.
* @param _model - exact model id passed to {@link GenerateOptions.model}.
* @param _signal - cancellation for this exact-model lookup; implementations
* must settle promptly after it aborts.
* @returns adapter-owned effort metadata, or `undefined` when unsupported.
*/
resolveModelReasoning(
_provider: string,
_model: string,
resolveModel(
provider: string,
model: string,
_signal?: AbortSignal,
): Promise<LlmModelReasoningInfo | undefined> {
return Promise.resolve(undefined)
): Promise<LlmResolvedModelInfo> {
return Promise.resolve({ provider, id: model, name: model })
}
/**
@@ -273,53 +258,59 @@ export class LlmService extends Service {
}
/**
* Resolve context capacity from the adapter that owns one exact route.
* This query is independent of the advisory model catalog: an unlisted model
* may return metadata, while `undefined` never rejects later routing.
* Resolve and validate all metadata from the adapter that owns one exact
* route. The result is detached from adapter-owned objects; catalog
* membership remains advisory and does not control request routing.
* @param provider - registered provider route to inspect.
* @param model - exact model id passed to the adapter.
* @returns detached context metadata, or `undefined` when the adapter has none.
* @param signal - optional cancellation for adapter-owned asynchronous lookup.
* @returns exact model identity plus available context and reasoning metadata.
*/
async resolveModelContext(
async resolveModelInfo(
provider: string,
model: string,
): Promise<LlmModelContext | undefined> {
const context = await this.registration(provider).adapter.resolveModelContext(provider, model)
if (context === undefined) return undefined
if (!Number.isInteger(context.contextWindow) || context.contextWindow <= 0) {
signal?: AbortSignal,
): Promise<LlmResolvedModelInfo> {
return this.resolveModelInfoFor(this.registration(provider), model, signal)
}
private async resolveModelInfoFor(
registration: AdapterRegistration,
model: string,
signal?: AbortSignal,
): Promise<LlmResolvedModelInfo> {
const provider = registration.provider.id
const resolved = await registration.adapter.resolveModel(provider, model, signal)
if (
typeof resolved.provider !== 'string'
|| resolved.provider !== provider
|| typeof resolved.id !== 'string'
|| resolved.id !== model
|| typeof resolved.name !== 'string'
|| resolved.name.length === 0
|| (resolved.description !== undefined && typeof resolved.description !== 'string')
) {
throw new LlmError(
`adapter returned invalid exact model metadata for provider "${provider}" model "${model}"`,
'INVALID_MODEL_INFO',
)
}
const context = resolved.context
if (context !== undefined && (!Number.isInteger(context.contextWindow) || context.contextWindow <= 0)) {
throw new LlmError(
`adapter returned invalid context metadata for provider "${provider}" model "${model}"`,
'INVALID_MODEL_CONTEXT',
)
}
return { contextWindow: context.contextWindow }
}
/**
* Resolve selectable reasoning efforts from the adapter that owns one exact
* route. Metadata is validated and detached; an absent result means an
* effort selector is unsupported for that model.
* @param provider - registered provider route to inspect.
* @param model - exact model id passed to the adapter.
* @param signal - optional cancellation for adapter-owned asynchronous lookup.
* @returns detached reasoning metadata, or `undefined` when unsupported.
*/
async resolveModelReasoning(
provider: string,
model: string,
signal?: AbortSignal,
): Promise<LlmModelReasoningInfo | undefined> {
return this.resolveModelReasoningFor(this.registration(provider), model, signal)
}
private async resolveModelReasoningFor(
registration: AdapterRegistration,
model: string,
signal?: AbortSignal,
): Promise<LlmModelReasoningInfo | undefined> {
const provider = registration.provider.id
const reasoning = await registration.adapter.resolveModelReasoning(provider, model, signal)
if (reasoning === undefined) return undefined
const info: LlmResolvedModelInfo = {
provider,
id: model,
name: resolved.name,
...resolved.description === undefined ? {} : { description: resolved.description },
...context === undefined ? {} : { context: { contextWindow: context.contextWindow } },
}
const reasoning = resolved.reasoning
if (reasoning === undefined) return info
if (reasoning.efforts.length === 0) {
throw new LlmError(
`adapter returned invalid reasoning metadata for provider "${provider}" model "${model}"`,
@@ -355,8 +346,11 @@ export class LlmService extends Service {
)
}
return {
efforts,
...reasoning.defaultEffort === undefined ? {} : { defaultEffort: reasoning.defaultEffort },
...info,
reasoning: {
efforts,
...reasoning.defaultEffort === undefined ? {} : { defaultEffort: reasoning.defaultEffort },
},
}
}
@@ -379,7 +373,7 @@ export class LlmService extends Service {
config: LlmCallConfig,
signal?: AbortSignal,
): Promise<LlmCallConfig> {
const reasoning = await this.resolveModelReasoningFor(registration, config.model, signal)
const reasoning = (await this.resolveModelInfoFor(registration, config.model, signal)).reasoning
const requested = config.reasoningEffort
if (reasoning === undefined) {
if (requested !== undefined) {
@@ -520,7 +514,7 @@ export class LlmService extends Service {
* `LlmError` with code `NO_ADAPTER` if no adapter is registered for
* `options.provider`. Replay state is retained only when the same adapter
* instance owns its historical provider and the target provider. Final
* adapter selection remains fixed through asynchronous reasoning resolution
* adapter selection remains fixed through asynchronous exact-model resolution
* and dispatch. Selection, dispatch, and iteration failures retain their
* original Error identity and are tagged in a call-local scope for narrow
* agent-loop request recovery; middleware and nested-call failures remain

View File

@@ -182,6 +182,14 @@ export interface LlmModelReasoningInfo {
defaultEffort?: ReasoningEffortId
}
/** Exact-route model metadata resolved by its owning adapter. */
export interface LlmResolvedModelInfo extends LlmModelInfo {
/** Provider-owned context capacity when known. */
context?: LlmModelContext
/** Adapter-owned selectable reasoning levels when exposed. */
reasoning?: LlmModelReasoningInfo
}
/**
* Raw streaming protocol emitted by adapters.
* Block indexes correlate interleaved deltas, and `block-end` carries the

View File

@@ -19,6 +19,7 @@ import type {
LlmModelInfo,
LlmModelReasoningInfo,
LlmProviderInfo,
LlmResolvedModelInfo,
} from '@deepseek-ai/dsh-llm'
class ScriptedAdapter extends LlmAdapter {
@@ -68,18 +69,17 @@ class CatalogAdapter extends ScriptedAdapter {
return Promise.resolve(this.models)
}
override resolveModelContext(
_provider: string,
override resolveModel(
provider: string,
model: string,
): Promise<LlmModelContext | undefined> {
return Promise.resolve(this.contexts[model])
}
override resolveModelReasoning(
_provider: string,
model: string,
): Promise<LlmModelReasoningInfo | undefined> {
return Promise.resolve(this.reasoning[model])
): Promise<LlmResolvedModelInfo> {
return Promise.resolve({
provider,
id: model,
name: model,
...this.contexts[model] === undefined ? {} : { context: this.contexts[model] },
...this.reasoning[model] === undefined ? {} : { reasoning: this.reasoning[model] },
})
}
}
@@ -667,8 +667,32 @@ describe('LlmService', () => {
expect(ctx.llm.listProviders()).toEqual([{ id: 'plain', name: 'plain' }])
await expect(ctx.llm.listModels('plain')).resolves.toEqual([])
await expect(ctx.llm.listModels('missing')).rejects.toMatchObject({ code: 'NO_ADAPTER' })
await expect(ctx.llm.resolveModelContext('plain', 'unlisted')).resolves.toBeUndefined()
await expect(ctx.llm.resolveModelContext('missing', 'm')).rejects.toMatchObject({ code: 'NO_ADAPTER' })
await expect(ctx.llm.resolveModelInfo('plain', 'unlisted')).resolves.toEqual({
provider: 'plain', id: 'unlisted', name: 'unlisted',
})
await expect(ctx.llm.resolveModelInfo('missing', 'm')).rejects.toMatchObject({ code: 'NO_ADAPTER' })
})
it.each([
[{ provider: 1, id: 'model', name: 'Model' }, 'non-string provider'],
[{ provider: 'other', id: 'model', name: 'Model' }, 'mismatched provider'],
[{ provider: 'route', id: 1, name: 'Model' }, 'non-string id'],
[{ provider: 'route', id: 'other', name: 'Model' }, 'mismatched id'],
[{ provider: 'route', id: 'model', name: 1 }, 'non-string name'],
[{ provider: 'route', id: 'model', name: '' }, 'empty name'],
[{ provider: 'route', id: 'model', name: 'Model', description: 1 }, 'non-string description'],
] as const)('rejects invalid exact model metadata (%s: %s)', async (metadata, _label) => {
const ctx = new Context()
await ctx.plugin(LlmService)
const adapter = new class extends ScriptedAdapter {
override resolveModel(): Promise<LlmResolvedModelInfo> {
return Promise.resolve(metadata as unknown as LlmResolvedModelInfo)
}
}(SCRIPT)
ctx.llm.registerAdapter(['route'], adapter)
await expect(ctx.llm.resolveModelInfo('route', 'model'))
.rejects.toMatchObject({ code: 'INVALID_MODEL_INFO' })
})
it('resolves detached model context independently of advisory catalog membership', async () => {
@@ -681,11 +705,13 @@ describe('LlmService', () => {
{ unlisted: source },
))
const resolved = await ctx.llm.resolveModelContext('route', 'unlisted')
expect(resolved).toEqual({ contextWindow: 32_000 })
const resolved = await ctx.llm.resolveModelInfo('route', 'unlisted')
expect(resolved.context).toEqual({ contextWindow: 32_000 })
source.contextWindow = 64_000
expect(resolved).toEqual({ contextWindow: 32_000 })
await expect(ctx.llm.resolveModelContext('route', 'other')).resolves.toBeUndefined()
expect(resolved.context).toEqual({ contextWindow: 32_000 })
await expect(ctx.llm.resolveModelInfo('route', 'other')).resolves.toEqual({
provider: 'route', id: 'other', name: 'other',
})
})
it('resolves detached adapter-owned reasoning metadata and materializes its default', async () => {
@@ -705,10 +731,10 @@ describe('LlmService', () => {
{ model: source },
))
const resolved = await ctx.llm.resolveModelReasoning('route', 'model')
expect(resolved).toEqual(source)
const resolved = await ctx.llm.resolveModelInfo('route', 'model')
expect(resolved.reasoning).toEqual(source)
source.efforts[0]!.name = 'mutated'
expect(resolved?.efforts[0]?.name).toBe('Standard')
expect(resolved.reasoning?.efforts[0]?.name).toBe('Standard')
await expect(ctx.llm.resolveCallConfig({ provider: 'route', model: 'model' })).resolves.toEqual({
provider: 'route',
model: 'model',
@@ -734,7 +760,7 @@ describe('LlmService', () => {
{},
{ model: metadata as unknown as LlmModelReasoningInfo },
))
await expect(ctx.llm.resolveModelReasoning('route', 'model'))
await expect(ctx.llm.resolveModelInfo('route', 'model'))
.rejects.toMatchObject({ code: 'INVALID_MODEL_REASONING' })
})
@@ -764,13 +790,16 @@ describe('LlmService', () => {
const ctx = new Context()
await ctx.plugin(LlmService)
const adapter = new class extends RecordingAdapter {
override resolveModelReasoning(
_provider: string,
_model: string,
): Promise<LlmModelReasoningInfo> {
return Promise.resolve({
override resolveModel(provider: string, model: string): Promise<LlmResolvedModelInfo> {
const reasoning: LlmModelReasoningInfo = {
efforts: [{ id: ReasoningEffortId('standard'), name: 'Standard' }],
defaultEffort: ReasoningEffortId('standard'),
}
return Promise.resolve({
provider,
id: model,
name: model,
reasoning,
})
}
}(SCRIPT)
@@ -799,19 +828,24 @@ describe('LlmService', () => {
expect(Object.isFrozen(adapter.lastOptions)).toBe(true)
})
it('pins one adapter registration across asynchronous reasoning resolution and dispatch', async () => {
it('pins one adapter registration across asynchronous exact-model resolution and dispatch', async () => {
const ctx = new Context()
await ctx.plugin(LlmService)
const started = Promise.withResolvers<undefined>()
const reasoning = Promise.withResolvers<LlmModelReasoningInfo>()
const first = new class extends RecordingAdapter {
override resolveModelReasoning(
_provider: string,
_model: string,
override async resolveModel(
provider: string,
model: string,
_signal?: AbortSignal,
): Promise<LlmModelReasoningInfo> {
): Promise<LlmResolvedModelInfo> {
started.resolve(undefined)
return reasoning.promise
return {
provider,
id: model,
name: model,
reasoning: await reasoning.promise,
}
}
}(SCRIPT)
const disposeFirst = ctx.llm.registerAdapter(['route'], first)
@@ -869,18 +903,18 @@ describe('LlmService', () => {
})).toThrow(expect.objectContaining({ code: 'INVALID_PREPARED_CALL' }))
})
it('passes cancellation through reasoning capability resolution', async () => {
it('passes cancellation through exact-model resolution', async () => {
const ctx = new Context()
await ctx.plugin(LlmService)
const started = Promise.withResolvers<undefined>()
const adapter = new class extends ScriptedAdapter {
override resolveModelReasoning(
override resolveModel(
_provider: string,
_model: string,
signal?: AbortSignal,
): Promise<LlmModelReasoningInfo | undefined> {
): Promise<LlmResolvedModelInfo> {
started.resolve(undefined)
return new Promise((_resolve, reject) => {
return new Promise<LlmResolvedModelInfo>((_resolve, reject) => {
if (signal === undefined) {
reject(new Error('missing reasoning signal'))
return
@@ -918,7 +952,7 @@ describe('LlmService', () => {
[],
{ model: { contextWindow } },
))
await expect(ctx.llm.resolveModelContext('route', 'model'))
await expect(ctx.llm.resolveModelInfo('route', 'model'))
.rejects.toMatchObject({ code: 'INVALID_MODEL_CONTEXT' })
},
)

View File

@@ -4,7 +4,7 @@ Replay-aware token measurement through the singleton `ctx.tokenMeter` service. I
## Configuration
The estimator has no settings. It intentionally uses one fixed heuristic: four characters per token plus structural overhead for roles, blocks, and request-envelope fields. Any key is rejected, including the obsolete global `contextWindow`; model capacity belongs to the adapter that owns an exact provider/model route and is available through `ctx.llm.resolveModelContext()`.
The estimator has no settings. It intentionally uses one fixed heuristic: four characters per token plus structural overhead for roles, blocks, and request-envelope fields. Any key is rejected, including the obsolete global `contextWindow`; model capacity belongs to the adapter that owns an exact provider/model route and is available through `ctx.llm.resolveModelInfo().context`.
## Measurement contract