feat(llm): interrogate a draft provider endpoint for its models
Once a pi-ai route became a declaration rather than a catalog lookup, adding an OpenAI-compatible gateway meant knowing its model ids up front. Most such endpoints publish that list at `GET /models`, but no seam operation could ask: every one is keyed by a registered provider route, and the provider being added has no route, no stored profile, and no stored credential — the endpoint and key are values in a form. Interrogation is therefore keyed by settings namespace, which a configuration surface already holds from the configurable-provider directory. `registerModelDiscovery` offers it per namespace, `discoverModels` asks, and the request carries the draft itself. The reply is candidates, not a catalog: every field but the id is optional because most listings disclose nothing else, and adopting one is a settings write like any other. Nothing here reads or writes settings or credentials, so `settings.yaml` still decides what a route serves. `llm.discoverModels` carries the same draft over the wire. Its apiKey is the third and last payload a secret may ride, and it is never stored, logged, or echoed; every refusal folds into `model-discovery-failed`, naming the endpoint asked but never the credential offered. The pi-ai side is a plain GET for OpenAI-compatible protocols only — their listing shape is the one gateways, self-hosted servers, and the official endpoints agree on. Others say so, sending the user to hand-entry rather than reporting a guessed shape as an empty provider. The reply is read under a four-megabyte ceiling held on the bytes actually received, because the endpoint is a URL the user typed.
This commit is contained in:
207
packages/llm/llm-pi-ai/src/discovery.ts
Normal file
207
packages/llm/llm-pi-ai/src/discovery.ts
Normal file
@@ -0,0 +1,207 @@
|
||||
/**
|
||||
* One-shot interrogation of a provider endpoint's model listing, serving the
|
||||
* configuration surface's "fetch available models" action.
|
||||
*
|
||||
* This is deliberately *not* a catalog refresh. Nothing here is stored: the
|
||||
* request carries a draft the user is still editing — an endpoint and a
|
||||
* credential neither of which may exist in `settings.yaml` yet — and the reply
|
||||
* is candidate metadata the surface offers for adoption. `settings.yaml`
|
||||
* remains the only thing that decides what a route serves.
|
||||
*
|
||||
* Only OpenAI-compatible protocols are interrogated. Their listing is the one
|
||||
* shape a gateway, a self-hosted server, and the official endpoints all agree
|
||||
* on, which is the case this action exists for; every other protocol reports
|
||||
* that it cannot be interrogated so the surface falls back to hand-entry
|
||||
* rather than guessing a response shape.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/discovery
|
||||
*/
|
||||
|
||||
import { LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import type { LlmDiscoveredModel, LlmModelDiscoveryRequest } from '@deepseek-ai/dsh-llm'
|
||||
import { attributionHeaders } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
/**
|
||||
* Protocols whose model listing this module can read. Every entry speaks
|
||||
* OpenAI's `GET /models` shape; pi-ai's other protocols are absent because a
|
||||
* wrong guess at their response shape would be reported as an empty provider
|
||||
* rather than as the gap it is.
|
||||
*/
|
||||
const LISTABLE_PROTOCOLS: ReadonlySet<string> = new Set([
|
||||
'azure-openai-responses',
|
||||
'openai-codex-responses',
|
||||
'openai-completions',
|
||||
'openai-responses',
|
||||
])
|
||||
|
||||
/**
|
||||
* Endpoint replies larger than this are refused. The endpoint is whatever URL
|
||||
* the user typed, so the ceiling holds on the bytes actually read rather than
|
||||
* on the length the server claims — the same two-stage shape `dsh-web-fetch`
|
||||
* uses for its own caller-supplied URLs, except that a truncated model listing
|
||||
* is not parseable, so overflow rejects instead of truncating.
|
||||
*/
|
||||
const MAX_RESPONSE_BYTES = 4 * 1024 * 1024
|
||||
|
||||
/** One entry of an OpenAI-compatible `GET /models` reply. */
|
||||
interface ListingEntry {
|
||||
id?: unknown
|
||||
/** Common gateway extensions; absent from the official listings. */
|
||||
name?: unknown
|
||||
display_name?: unknown
|
||||
context_window?: unknown
|
||||
context_length?: unknown
|
||||
max_tokens?: unknown
|
||||
max_output_tokens?: unknown
|
||||
}
|
||||
|
||||
/** A positive integer field of a listing entry, or `undefined` when absent or unusable. */
|
||||
function capacity(...candidates: readonly unknown[]): number | undefined {
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === 'number' && Number.isInteger(candidate) && candidate > 0) return candidate
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/** A non-empty string field of a listing entry, or `undefined`. */
|
||||
function label(...candidates: readonly unknown[]): string | undefined {
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === 'string' && candidate.length > 0) return candidate
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Join the endpoint base with the listing path. The base is treated as a
|
||||
* prefix rather than a URL to resolve against, so a deployment path such as
|
||||
* `https://gateway.example/openai/v1` keeps its segments instead of losing
|
||||
* them to `URL` resolution.
|
||||
*/
|
||||
function listingUrl(baseURL: string): string {
|
||||
return `${baseURL.replace(/\/+$/, '')}/models`
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a reply body, refusing one that outgrows the ceiling. A declared length
|
||||
* is checked first so an honest server is turned away without transferring
|
||||
* anything; the accumulated total is what actually enforces the bound, because
|
||||
* a server that under-declares (or streams) tells us nothing up front.
|
||||
*/
|
||||
async function readBounded(response: Response, url: string): Promise<string> {
|
||||
const oversized = (): LlmError =>
|
||||
new LlmError(`${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, 'DISCOVERY_FAILED')
|
||||
const declared = Number(response.headers.get('content-length') ?? Number.NaN)
|
||||
if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
|
||||
await response.body?.cancel()
|
||||
throw oversized()
|
||||
}
|
||||
/* v8 ignore next -- fetch always exposes a body stream on a 2xx Response; the null guard is defensive. */
|
||||
if (response.body === null) return ''
|
||||
const reader = response.body.getReader()
|
||||
const chunks: Uint8Array[] = []
|
||||
let total = 0
|
||||
try {
|
||||
for (;;) {
|
||||
const { done, value } = await reader.read()
|
||||
if (done) break
|
||||
total += value.byteLength
|
||||
if (total > MAX_RESPONSE_BYTES) throw oversized()
|
||||
chunks.push(value)
|
||||
}
|
||||
} finally {
|
||||
/* v8 ignore next 4 -- cancel() after a completed or abandoned read settles without rejecting; unobserved best-effort cleanup. */
|
||||
await reader.cancel().catch(() => {
|
||||
// Cancel after a drained read, or after this function walked away from
|
||||
// an oversized one, is cleanup; the reply is already decided either way.
|
||||
})
|
||||
}
|
||||
const body = new Uint8Array(total)
|
||||
let offset = 0
|
||||
for (const chunk of chunks) {
|
||||
body.set(chunk, offset)
|
||||
offset += chunk.byteLength
|
||||
}
|
||||
return new TextDecoder().decode(body)
|
||||
}
|
||||
|
||||
/**
|
||||
* Read one OpenAI-compatible listing reply. Entries without a usable id are
|
||||
* skipped rather than failing the whole interrogation: a single malformed row
|
||||
* should not deny the user the rest of a working endpoint's catalog.
|
||||
*/
|
||||
function readListing(body: unknown): LlmDiscoveredModel[] {
|
||||
const data = (body as { data?: unknown } | null)?.data
|
||||
if (!Array.isArray(data)) {
|
||||
throw new LlmError(
|
||||
'the endpoint\'s model listing has no "data" array; enter this provider\'s models by hand',
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
const models: LlmDiscoveredModel[] = []
|
||||
for (const raw of data) {
|
||||
const entry = raw as ListingEntry | null
|
||||
const id = label(entry?.id)
|
||||
if (id === undefined) continue
|
||||
const name = label(entry?.name, entry?.display_name)
|
||||
const contextWindow = capacity(entry?.context_window, entry?.context_length)
|
||||
const maxTokens = capacity(entry?.max_output_tokens, entry?.max_tokens)
|
||||
models.push({
|
||||
id,
|
||||
...name === undefined ? {} : { name },
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
...maxTokens === undefined ? {} : { maxTokens },
|
||||
})
|
||||
}
|
||||
return models
|
||||
}
|
||||
|
||||
/**
|
||||
* Interrogate one draft provider endpoint for the models it advertises.
|
||||
* @param request - the endpoint, protocol, and one-shot credential to use.
|
||||
* @returns the advertised models in endpoint order.
|
||||
* @throws LlmError when the protocol has no readable listing, the endpoint
|
||||
* refuses or fails the request, or the reply is not a model listing.
|
||||
*/
|
||||
export async function discoverModels(
|
||||
request: LlmModelDiscoveryRequest,
|
||||
): Promise<readonly LlmDiscoveredModel[]> {
|
||||
const api = request.api ?? 'openai-completions'
|
||||
if (!LISTABLE_PROTOCOLS.has(api)) {
|
||||
throw new LlmError(
|
||||
`pi-ai protocol "${api}" has no model listing this build can read; enter this provider's models by hand`,
|
||||
'DISCOVERY_UNSUPPORTED',
|
||||
)
|
||||
}
|
||||
const url = listingUrl(request.baseURL)
|
||||
let response: Response
|
||||
try {
|
||||
response = await fetch(url, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
accept: 'application/json',
|
||||
...request.apiKey === undefined ? {} : { authorization: `Bearer ${request.apiKey}` },
|
||||
...attributionHeaders(),
|
||||
},
|
||||
...request.signal === undefined ? {} : { signal: request.signal },
|
||||
})
|
||||
} catch (error: unknown) {
|
||||
if (request.signal?.aborted) {
|
||||
throw new LlmError('model discovery aborted by caller', 'ABORTED', { cause: error })
|
||||
}
|
||||
throw new LlmError(`could not reach ${url}`, 'DISCOVERY_FAILED', { cause: error })
|
||||
}
|
||||
if (!response.ok) {
|
||||
throw new LlmError(
|
||||
`${url} answered ${response.status}${response.status === 401 || response.status === 403 ? '; check the API key' : ''}`,
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
const text = await readBounded(response, url)
|
||||
let body: unknown
|
||||
try {
|
||||
body = JSON.parse(text)
|
||||
} catch (error: unknown) {
|
||||
throw new LlmError(`${url} did not answer with JSON`, 'DISCOVERY_FAILED', { cause: error })
|
||||
}
|
||||
return readListing(body)
|
||||
}
|
||||
@@ -50,6 +50,7 @@ import { PiAiAdapter } from './adapter.ts'
|
||||
import { catalogProviderIds } from './catalog.ts'
|
||||
import { assertServiceable, Config, resolveProfiles } from './config.ts'
|
||||
import type { ResolvedPiAiProviderProfile } from './config.ts'
|
||||
import { discoverModels } from './discovery.ts'
|
||||
|
||||
export { PiAiAdapter } from './adapter.ts'
|
||||
export type { PiAiAdapterOptions } from './adapter.ts'
|
||||
@@ -176,6 +177,10 @@ export function apply(ctx: Context, config: Config): void {
|
||||
directoryFacts = entries
|
||||
}
|
||||
ensureDirectory()
|
||||
// Interrogating an endpoint is a configuration-time action over a draft, so
|
||||
// it is offered for the whole namespace rather than per route: the provider
|
||||
// a surface is adding does not exist yet.
|
||||
ctx.llm.registerModelDiscovery(NS, discoverModels)
|
||||
// Route effects bind to this apply fiber via the stable `ctx` reference,
|
||||
// even when a swap runs inside the scoped settings callback below. A bare
|
||||
// mount (zero routes) is the dormant posture: nothing registers until a
|
||||
|
||||
211
packages/llm/llm-pi-ai/tests/discovery.spec.ts
Normal file
211
packages/llm/llm-pi-ai/tests/discovery.spec.ts
Normal file
@@ -0,0 +1,211 @@
|
||||
import { createServer } from 'node:http'
|
||||
import type { IncomingMessage, Server, ServerResponse } from 'node:http'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import LlmService, { userAgent } from '@deepseek-ai/dsh-llm'
|
||||
import * as LlmPiAi from '@deepseek-ai/dsh-llm-pi-ai'
|
||||
|
||||
const servers: Server[] = []
|
||||
|
||||
afterEach(async () => {
|
||||
await Promise.all(servers.splice(0).map(server => new Promise(resolve => server.close(resolve))))
|
||||
})
|
||||
|
||||
interface ListingServer {
|
||||
url: string
|
||||
paths: string[]
|
||||
headers: IncomingMessage['headers'][]
|
||||
}
|
||||
|
||||
/**
|
||||
* A stand-in provider that answers one scripted `GET /models`. `chunks` writes
|
||||
* without a declared length, which is how a real streamed reply arrives.
|
||||
*/
|
||||
async function listingServer(behavior: {
|
||||
status?: number
|
||||
body?: string
|
||||
chunks?: string[]
|
||||
}): Promise<ListingServer> {
|
||||
const paths: string[] = []
|
||||
const headers: IncomingMessage['headers'][] = []
|
||||
const server = createServer((request: IncomingMessage, response: ServerResponse) => {
|
||||
paths.push(request.url ?? '')
|
||||
headers.push(request.headers)
|
||||
if (behavior.chunks !== undefined) {
|
||||
// No declared length: the ceiling has to hold on what is read.
|
||||
response.writeHead(behavior.status ?? 200, { 'content-type': 'application/json' })
|
||||
for (const chunk of behavior.chunks) response.write(chunk)
|
||||
response.end()
|
||||
return
|
||||
}
|
||||
const body = behavior.body ?? '{}'
|
||||
response.writeHead(behavior.status ?? 200, {
|
||||
'content-type': 'application/json',
|
||||
'content-length': String(Buffer.byteLength(body)),
|
||||
})
|
||||
response.end(body)
|
||||
})
|
||||
servers.push(server)
|
||||
await new Promise<void>(resolve => server.listen(0, '127.0.0.1', resolve))
|
||||
const address = server.address()
|
||||
if (address === null || typeof address === 'string') throw new Error('no port')
|
||||
return { url: `http://127.0.0.1:${address.port}`, paths, headers }
|
||||
}
|
||||
|
||||
/** A bare dormant mount: discovery is offered whether or not a route exists. */
|
||||
async function harness(): Promise<Context> {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
await ctx.plugin(LlmPiAi, {})
|
||||
return ctx
|
||||
}
|
||||
|
||||
describe('draft-provider model discovery', () => {
|
||||
it('reads an OpenAI-compatible listing and keeps the capacities it discloses', async () => {
|
||||
const server = await listingServer({
|
||||
body: JSON.stringify({
|
||||
data: [
|
||||
{ id: 'acme-large', display_name: 'Acme Large', context_length: 65_536, max_output_tokens: 4096 },
|
||||
{ id: 'acme-small' },
|
||||
],
|
||||
}),
|
||||
})
|
||||
const ctx = await harness()
|
||||
|
||||
const models = await ctx.llm.discoverModels('llm-pi-ai', { baseURL: `${server.url}/v1`, apiKey: 'probe-key' })
|
||||
|
||||
expect(models).toEqual([
|
||||
{ id: 'acme-large', name: 'Acme Large', contextWindow: 65_536, maxTokens: 4096 },
|
||||
{ id: 'acme-small' },
|
||||
])
|
||||
expect(server.paths).toEqual(['/v1/models'])
|
||||
expect(server.headers[0]?.authorization).toBe('Bearer probe-key')
|
||||
expect(server.headers[0]?.['user-agent']).toBe(userAgent())
|
||||
})
|
||||
|
||||
it('keeps a deployment path instead of resolving it away', async () => {
|
||||
const server = await listingServer({ body: JSON.stringify({ data: [{ id: 'm' }] }) })
|
||||
const ctx = await harness()
|
||||
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { baseURL: `${server.url}/openai/v1/` })
|
||||
|
||||
expect(server.paths).toEqual(['/openai/v1/models'])
|
||||
})
|
||||
|
||||
it('offers no credential when the draft names none', async () => {
|
||||
const server = await listingServer({ body: JSON.stringify({ data: [{ id: 'm' }] }) })
|
||||
const ctx = await harness()
|
||||
|
||||
await ctx.llm.discoverModels('llm-pi-ai', { baseURL: server.url })
|
||||
|
||||
expect(server.headers[0]?.authorization).toBeUndefined()
|
||||
})
|
||||
|
||||
it('drops unusable rows rather than failing the whole listing', async () => {
|
||||
const server = await listingServer({
|
||||
body: JSON.stringify({
|
||||
data: [
|
||||
{ id: 'good' },
|
||||
{ id: '' },
|
||||
{ name: 'no id at all' },
|
||||
null,
|
||||
{ id: 'good' },
|
||||
{ id: 'zero-capacity', context_length: 0, max_tokens: -1 },
|
||||
],
|
||||
}),
|
||||
})
|
||||
const ctx = await harness()
|
||||
|
||||
expect(await ctx.llm.discoverModels('llm-pi-ai', { baseURL: server.url }))
|
||||
.toEqual([{ id: 'good' }, { id: 'zero-capacity' }])
|
||||
})
|
||||
|
||||
it('points at the credential for a rejected one, and only then', async () => {
|
||||
const ctx = await harness()
|
||||
|
||||
for (const status of [401, 403]) {
|
||||
const refused = await listingServer({ status, body: '{"error":"nope"}' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: refused.url, apiKey: 'wrong' }))
|
||||
.rejects.toThrow(new RegExp(`answered ${status}; check the API key`))
|
||||
}
|
||||
|
||||
// A server fault is not a credential problem, so it must not send the user
|
||||
// off to re-check a key that is fine.
|
||||
const broken = await listingServer({ status: 500, body: '{"error":"boom"}' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: broken.url, apiKey: 'fine' }))
|
||||
.rejects.toThrow(/answered 500$/)
|
||||
})
|
||||
|
||||
it('reports a reply that is not a model listing', async () => {
|
||||
const server = await listingServer({ body: '{"models":[]}' })
|
||||
const ctx = await harness()
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: server.url }))
|
||||
.rejects.toThrow(/no "data" array; enter this provider's models by hand/)
|
||||
|
||||
const broken = await listingServer({ body: 'not json at all' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: broken.url }))
|
||||
.rejects.toThrow(/did not answer with JSON/)
|
||||
})
|
||||
|
||||
it('refuses an oversized reply, whether its length is declared or streamed', async () => {
|
||||
const ctx = await harness()
|
||||
// Just over the four-megabyte ceiling, as one padded model row.
|
||||
const oversized = `{"data":[{"id":"m","pad":"${'x'.repeat(4 * 1024 * 1024)}"}]}`
|
||||
|
||||
const declared = await listingServer({ body: oversized })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: declared.url }))
|
||||
.rejects.toThrow(/answered with more than 4194304 bytes/)
|
||||
|
||||
// A streamed reply declares no length, so the ceiling has to hold on the
|
||||
// body the harness actually read.
|
||||
const streamed = await listingServer({ chunks: ['{"data":[{"id":"m","pad":"', 'x'.repeat(4 * 1024 * 1024), '"}]}'] })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: streamed.url }))
|
||||
.rejects.toThrow(/answered with more than 4194304 bytes/)
|
||||
})
|
||||
|
||||
it('reports an unreachable endpoint instead of an empty catalog', async () => {
|
||||
const ctx = await harness()
|
||||
// Port 9 is the discard service: nothing accepts a connection there.
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: 'http://127.0.0.1:9/v1' }))
|
||||
.rejects.toMatchObject({ code: 'DISCOVERY_FAILED' })
|
||||
})
|
||||
|
||||
it('says which protocols it cannot interrogate rather than guessing a shape', async () => {
|
||||
const ctx = await harness()
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', {
|
||||
baseURL: 'https://gateway.example/v1',
|
||||
api: 'anthropic-messages',
|
||||
})).rejects.toMatchObject({ code: 'DISCOVERY_UNSUPPORTED' })
|
||||
})
|
||||
|
||||
it('honors caller cancellation', async () => {
|
||||
const ctx = await harness()
|
||||
const aborted = AbortSignal.abort('test cancellation')
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', {
|
||||
baseURL: 'http://127.0.0.1:9/v1',
|
||||
signal: aborted,
|
||||
})).rejects.toMatchObject({ code: 'ABORTED' })
|
||||
})
|
||||
|
||||
it('is offered for the namespace, and refuses one it does not serve', async () => {
|
||||
const ctx = await harness()
|
||||
|
||||
expect(ctx.llm.listModelDiscoveryNamespaces()).toEqual(['llm-pi-ai'])
|
||||
await expect(ctx.llm.discoverModels('llm-deepseek', { baseURL: 'https://api.deepseek.com' }))
|
||||
.rejects.toMatchObject({ code: 'NO_DISCOVERY' })
|
||||
await expect(ctx.llm.discoverModels('llm-pi-ai', { baseURL: '' }))
|
||||
.rejects.toMatchObject({ code: 'INVALID_DISCOVERY' })
|
||||
})
|
||||
|
||||
it('withdraws the offer when the plugin unloads', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const fiber = await ctx.plugin(LlmPiAi, {})
|
||||
expect(ctx.llm.listModelDiscoveryNamespaces()).toEqual(['llm-pi-ai'])
|
||||
|
||||
await fiber.dispose()
|
||||
|
||||
expect(ctx.llm.listModelDiscoveryNamespaces()).toEqual([])
|
||||
})
|
||||
})
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/llm/README.md
|
||||
README.md: e09ec685ed0ab1e2492749237c277a874eb3b246
|
||||
README.zh.md: ca98e875a90eb16e32bc405d77cd5b2b56644180
|
||||
README.md: 60cc94b6375030955136b4efaf969b69bca2530a
|
||||
README.zh.md: 5b24a1e311c37d13dc4f287e5ae4b57efe00e4e6
|
||||
|
||||
@@ -14,6 +14,9 @@ An adapter registry plus a single streaming call surface, interceptable via a wa
|
||||
- `ctx.llm.listProviders(): LlmProviderInfo[]` Describe registered provider routes in registration order.
|
||||
- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle` Declare provider routes an adapter plugin can activate through configuration — registered or dormant — each naming its owning settings namespace and the path to its profile inside that section. All-or-nothing (`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`), disposed with the calling fiber. The handle also carries `replace(entries)`: the candidate set is validated in full before anything moves, so an entry another registration already declares leaves the current set intact, and an empty array is legal there. A plugin whose declared set follows its configuration must use `replace` rather than disposing and re-registering — the latter strands the directory empty whenever the new set is refused.
|
||||
- `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` List the declared directory in declaration order; configuration surfaces merge it with `listProviders()` to mark each entry live or dormant.
|
||||
- `ctx.llm.registerModelDiscovery(settingsNs: string, discover): () => void` Offer to interrogate provider endpoints for the settings namespace this plugin owns. One offer per namespace (`INVALID_DISCOVERY`/`DUPLICATE_DISCOVERY`), disposed with the calling fiber.
|
||||
- `ctx.llm.listModelDiscoveryNamespaces(): string[]` List the namespaces that can interrogate an endpoint, so a surface offers the action only where it works.
|
||||
- `ctx.llm.discoverModels(settingsNs: string, request: LlmModelDiscoveryRequest): Promise<LlmDiscoveredModel[]>` Ask one endpoint which models it advertises.
|
||||
- `ctx.llm.providerRetryPolicy(provider: string): ResolvedRetryPolicy` Return the provider-owned retry policy captured during registration, with normal defaults resolved.
|
||||
- `ctx.llm.listModels(provider: string): Promise<LlmModelInfo[]>` Discover the models one registered provider currently advertises.
|
||||
- `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<LlmResolvedModelInfo>` Resolve validated exact-model identity plus available context, output-default, and reasoning metadata from the owning adapter, with optional cancellation for asynchronous adapters.
|
||||
@@ -23,6 +26,8 @@ An adapter registry plus a single streaming call surface, interceptable via a wa
|
||||
|
||||
`LlmService` preserves errors from final adapter selection, synchronous dispatch, iterator construction, and iteration, and binds their provenance to the exact stream handle returned for that model call. `isLlmAdapterFailure(stream, value)` reports only errors from that call's final adapter boundary; `llmFailureOf(stream, value)` returns the adjacent immutable `LlmFailure`; `llmRetryPolicyOf(stream)` returns the immutable policy of the exact registration selected at that boundary, even if the route is later disposed or replaced. A call that never reaches a final adapter has no serving policy. Nested model calls, `llm/stream` middleware, and downstream consumer failures remain unclassified for the outer call. Classification never replaces or mutates the adapter's original coded `Error`.
|
||||
|
||||
Interrogating an endpoint is configuration-time work over a *draft*, which is why it is keyed by settings namespace rather than by provider route: the provider a surface is adding does not exist yet, so there is no route to name. The request carries the endpoint, the protocol, and a credential the harness uses for that one interrogation and never stores — nothing here reads or writes settings or credentials, and the reply is candidate metadata a surface may offer for adoption, never a registered catalog. `LlmDiscoveredModel` makes every field but `id` optional because most provider listings disclose an id and nothing else; a surface adopting one still owes the capacities its adapter requires. Duplicate and unusable ids are dropped, an unserved namespace fails with `NO_DISCOVERY`, and an empty namespace or endpoint fails with `INVALID_DISCOVERY`.
|
||||
|
||||
Provider and model metadata is a discovery surface, not a routing whitelist. `registerAdapter()` still owns provider exclusivity and captures the adapter's retry policy for each route, while an adapter may accept model ids absent from `listModels()`; consumers must not reject a request because its model is unlisted. Returned selector metadata is detached and invalid or duplicate adapter entries fail with `INVALID_ADAPTER` or `INVALID_CATALOG`.
|
||||
|
||||
Every topology commit point — adapter routes registering or disposing, directory entries appearing or withdrawing — emits the payload-free `llm/adapters-updated` event after the mutation, so consumers re-read `listProviders()`/`listModels()`/`listConfigurableProviders()` instead of polling. Observer failures are contained (logged, non-vetoing); only `INVARIANT`-coded failures rethrow after the fan-out.
|
||||
|
||||
@@ -14,6 +14,9 @@
|
||||
- `ctx.llm.listProviders(): LlmProviderInfo[]` 按注册顺序描述已注册提供方路由。
|
||||
- `ctx.llm.registerConfigurableProviders(entries: readonly LlmConfigurableProvider[]): DirectoryRegistrationHandle` 声明适配器插件可通过配置激活的提供方路由——无论已注册还是休眠——每个条目指明其所属 settings namespace,以及 profile 在该分节内的路径。要么全部成功,要么全部不生效(`INVALID_DIRECTORY`/`DUPLICATE_DIRECTORY`),并随调用 fiber dispose。该句柄还带 `replace(entries)`:候选集合会先被整体校验,因此其中若有条目已被另一个注册声明,当前集合原封不动;此处允许传空数组。声明集合随配置变化的插件必须使用 `replace`,而不是先 dispose 再重新注册——后者会在新集合被拒时让目录整个落空。
|
||||
- `ctx.llm.listConfigurableProviders(): LlmConfigurableProvider[]` 按声明顺序列出已声明的目录;配置界面将其与 `listProviders()` 合并,为每个条目标注存活或休眠。
|
||||
- `ctx.llm.registerModelDiscovery(settingsNs: string, discover): () => void` 为本插件拥有的 settings namespace 提供「询问提供方端点」的能力。每个 namespace 只能有一个(`INVALID_DISCOVERY`/`DUPLICATE_DISCOVERY`),并随调用 fiber dispose。
|
||||
- `ctx.llm.listModelDiscoveryNamespaces(): string[]` 列出可以询问端点的 namespace,让界面只在可用之处提供该动作。
|
||||
- `ctx.llm.discoverModels(settingsNs: string, request: LlmModelDiscoveryRequest): Promise<LlmDiscoveredModel[]>` 询问某个端点它公布了哪些模型。
|
||||
- `ctx.llm.providerRetryPolicy(provider: string): ResolvedRetryPolicy` 返回注册时捕获的提供方重试策略,并解析 normal 默认值。
|
||||
- `ctx.llm.listModels(provider: string): Promise<LlmModelInfo[]>` 发现某个已注册提供方当前公布的模型。
|
||||
- `ctx.llm.resolveModelInfo(provider: string, model: string, signal?: AbortSignal): Promise<LlmResolvedModelInfo>` 从拥有精确路由的适配器解析经校验的确切模型身份,以及可用上下文、输出默认值和推理(reasoning)元数据;异步适配器可选地支持取消。
|
||||
@@ -23,6 +26,8 @@
|
||||
|
||||
`LlmService` 保留来自最终适配器选择、同步 dispatch、iterator 构造与迭代的错误,并将其溯源绑定到该次模型调用返回的精确流句柄。`isLlmAdapterFailure(stream, value)` 只报告该调用最终适配器边界的错误;`llmFailureOf(stream, value)` 返回关联的不可变 `LlmFailure`;`llmRetryPolicyOf(stream)` 返回在该边界选中的确切注册所对应的不可变策略,即使之后释放或替换路由也不变。未到达最终适配器的调用没有服务策略。嵌套模型调用、`llm/stream` middleware 和下游消费方失败对外层调用仍未分类。分类绝不替换或更改适配器原有的带代码 `Error`。
|
||||
|
||||
询问端点属于配置期针对**草稿**的操作,因此以 settings namespace 而非提供方路由为键:界面正在新增的提供方还不存在,也就没有路由可点名。请求携带端点、协议,以及一条 harness 只用于这一次询问、绝不存储的凭据——这里既不读也不写 settings 与 credentials,回复是界面可供用户采纳的候选元数据,而不是已注册的 catalog。`LlmDiscoveredModel` 除 `id` 外每个字段都是可选的,因为大多数提供方列表只公布 id;采纳其中一条的界面仍要补上其适配器所需的容量。重复与不可用的 id 会被丢弃,无人服务的 namespace 以 `NO_DISCOVERY` 失败,空 namespace 或空端点以 `INVALID_DISCOVERY` 失败。
|
||||
|
||||
提供方与模型元数据是发现接口,不是路由白名单。`registerAdapter()` 仍拥有提供方排他性,并为每条路由捕获适配器的重试策略;适配器则可以接受 `listModels()` 中不存在的模型 id,消费方禁止因模型未列出而拒绝请求。返回的 selector 元数据与输入脱离,无效或重复适配器配置项会以 `INVALID_ADAPTER` 或 `INVALID_CATALOG` 失败。
|
||||
|
||||
每个拓扑提交点——适配器路由注册或 dispose、目录条目出现或撤回——都会在变更之后发出无载荷的 `llm/adapters-updated` 事件,消费方因此重读 `listProviders()`/`listModels()`/`listConfigurableProviders()` 而非轮询。观察者故障会被隔离(记录日志、不否决);只有带 `INVARIANT` 码的故障会在扇出后重新抛出。
|
||||
|
||||
@@ -10,8 +10,10 @@ import { Context, Service } from 'cordis'
|
||||
import type {
|
||||
GenerateOptions,
|
||||
LlmConfigurableProvider,
|
||||
LlmDiscoveredModel,
|
||||
LlmFailure,
|
||||
LlmModelContext,
|
||||
LlmModelDiscoveryRequest,
|
||||
LlmModelInfo,
|
||||
LlmResolvedModelInfo,
|
||||
LlmProviderInfo,
|
||||
@@ -253,6 +255,10 @@ export interface DirectoryRegistrationHandle {
|
||||
export class LlmService extends Service {
|
||||
private adapters = new Map<string, AdapterRegistration>()
|
||||
private directory = new Map<string, LlmConfigurableProvider>()
|
||||
private discoveries = new Map<
|
||||
string,
|
||||
(request: LlmModelDiscoveryRequest) => Promise<readonly LlmDiscoveredModel[]>
|
||||
>()
|
||||
|
||||
constructor(ctx: Context) {
|
||||
super(ctx, 'llm')
|
||||
@@ -456,6 +462,80 @@ export class LlmService extends Service {
|
||||
return [...this.directory.values()].map(entry => ({ ...entry, settingsPath: [...entry.settingsPath] }))
|
||||
}
|
||||
|
||||
/**
|
||||
* Offer to interrogate provider endpoints on behalf of the settings
|
||||
* namespace this plugin owns. The namespace is the key because that is what
|
||||
* a configuration surface already holds from the configurable-provider
|
||||
* directory, and because a provider being *added* has no route to name yet.
|
||||
* Disposed with the fiber.
|
||||
* @param settingsNs - the namespace whose profiles this discovery serves.
|
||||
* @param discover - interrogates one endpoint; must honor `request.signal`.
|
||||
* @returns the disposer that withdraws the offer.
|
||||
*/
|
||||
registerModelDiscovery(
|
||||
settingsNs: string,
|
||||
discover: (request: LlmModelDiscoveryRequest) => Promise<readonly LlmDiscoveredModel[]>,
|
||||
): () => void {
|
||||
const dispose = this.ctx.effect(function* (this: LlmService) {
|
||||
if (settingsNs.length === 0) {
|
||||
throw new LlmError('model discovery needs a non-empty settings namespace', 'INVALID_DISCOVERY')
|
||||
}
|
||||
if (this.discoveries.has(settingsNs)) {
|
||||
throw new LlmError(`model discovery for "${settingsNs}" is already registered`, 'DUPLICATE_DISCOVERY')
|
||||
}
|
||||
this.discoveries.set(settingsNs, discover)
|
||||
yield () => {
|
||||
this.discoveries.delete(settingsNs)
|
||||
}
|
||||
}.bind(this), 'llm.registerModelDiscovery()')
|
||||
return () => void dispose()
|
||||
}
|
||||
|
||||
/**
|
||||
* List the settings namespaces that can interrogate a provider endpoint, so
|
||||
* a surface can offer the action only where it will work.
|
||||
* @returns the namespaces in registration order.
|
||||
*/
|
||||
listModelDiscoveryNamespaces(): string[] {
|
||||
return [...this.discoveries.keys()]
|
||||
}
|
||||
|
||||
/**
|
||||
* Interrogate one provider endpoint for the models it advertises. The
|
||||
* request describes a draft, not a stored route, so nothing here reads or
|
||||
* writes settings or credentials — the caller owns both, and the reply is
|
||||
* candidate metadata a surface may offer for adoption.
|
||||
* @param settingsNs - namespace whose registered discovery serves this draft.
|
||||
* @param request - the endpoint, protocol, and one-shot credential to use.
|
||||
* @returns the advertised models, deduplicated in endpoint order.
|
||||
*/
|
||||
async discoverModels(
|
||||
settingsNs: string,
|
||||
request: LlmModelDiscoveryRequest,
|
||||
): Promise<LlmDiscoveredModel[]> {
|
||||
const discover = this.discoveries.get(settingsNs)
|
||||
if (discover === undefined) {
|
||||
throw new LlmError(`no model discovery is registered for "${settingsNs}"`, 'NO_DISCOVERY')
|
||||
}
|
||||
if (request.baseURL.length === 0) {
|
||||
throw new LlmError('model discovery needs a non-empty baseURL', 'INVALID_DISCOVERY')
|
||||
}
|
||||
const discovered = await discover(request)
|
||||
const seen = new Set<string>()
|
||||
const models: LlmDiscoveredModel[] = []
|
||||
for (const model of discovered) {
|
||||
if (typeof model.id !== 'string' || model.id.length === 0 || seen.has(model.id)) continue
|
||||
seen.add(model.id)
|
||||
models.push({
|
||||
id: model.id,
|
||||
...model.name === undefined ? {} : { name: model.name },
|
||||
...model.contextWindow === undefined ? {} : { contextWindow: model.contextWindow },
|
||||
...model.maxTokens === undefined ? {} : { maxTokens: model.maxTokens },
|
||||
})
|
||||
}
|
||||
return models
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the retry policy captured when one provider route was registered.
|
||||
* @param provider - registered provider route to inspect.
|
||||
|
||||
@@ -139,6 +139,39 @@ export interface LlmConfigurableProvider {
|
||||
settingsPath: readonly string[]
|
||||
}
|
||||
|
||||
/**
|
||||
* One interrogation of a provider endpoint that configuration has not stored
|
||||
* yet. Configuration surfaces send the draft a user is still editing, so the
|
||||
* request carries the endpoint and credential directly instead of naming a
|
||||
* route: a provider being added has no route to name.
|
||||
*/
|
||||
export interface LlmModelDiscoveryRequest {
|
||||
/** Endpoint to interrogate. */
|
||||
baseURL: string
|
||||
/** Wire protocol the endpoint speaks, when the draft names one. */
|
||||
api?: string
|
||||
/** Credential for this interrogation alone; the harness never stores it. */
|
||||
apiKey?: string
|
||||
/** Caller cancellation; implementations must settle promptly after it aborts. */
|
||||
signal?: AbortSignal
|
||||
}
|
||||
|
||||
/**
|
||||
* One model an endpoint reports about itself. Every field but the id is
|
||||
* optional because most provider listings disclose an id and nothing else;
|
||||
* a surface adopting one of these still owes the capacities its adapter needs.
|
||||
*/
|
||||
export interface LlmDiscoveredModel {
|
||||
/** Model id the endpoint accepts. */
|
||||
id: string
|
||||
/** Human-readable name when the endpoint supplies one. */
|
||||
name?: string
|
||||
/** Maximum combined request and response context, when disclosed. */
|
||||
contextWindow?: number
|
||||
/** Maximum output tokens, when disclosed. */
|
||||
maxTokens?: number
|
||||
}
|
||||
|
||||
/** One adapter-discovered model; catalog membership is advisory, not request validation. */
|
||||
export interface LlmModelInfo {
|
||||
/** Provider route that owns this model entry. */
|
||||
|
||||
@@ -205,3 +205,55 @@ describe('configurable-provider directory', () => {
|
||||
expect(ctx.llm.listConfigurableProviders()).toHaveLength(1)
|
||||
})
|
||||
})
|
||||
|
||||
describe('model discovery registry', () => {
|
||||
it('offers one interrogation per settings namespace and disposes with its fiber', async () => {
|
||||
const ctx = await setup()
|
||||
const discover = vi.fn(() => Promise.resolve([{ id: 'from-endpoint' }]))
|
||||
|
||||
const dispose = ctx.llm.registerModelDiscovery('llm-example', discover)
|
||||
expect(ctx.llm.listModelDiscoveryNamespaces()).toEqual(['llm-example'])
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-example', { baseURL: 'https://gateway.example/v1' }))
|
||||
.resolves.toEqual([{ id: 'from-endpoint' }])
|
||||
expect(discover).toHaveBeenCalledWith({ baseURL: 'https://gateway.example/v1' })
|
||||
|
||||
dispose()
|
||||
expect(ctx.llm.listModelDiscoveryNamespaces()).toEqual([])
|
||||
})
|
||||
|
||||
it('rejects an unnamed namespace and a second registration of the same one', async () => {
|
||||
const ctx = await setup()
|
||||
const discover = (): Promise<never[]> => Promise.resolve([])
|
||||
|
||||
expect(() => ctx.llm.registerModelDiscovery('', discover)).toThrow(/non-empty settings namespace/)
|
||||
ctx.llm.registerModelDiscovery('llm-example', discover)
|
||||
expect(() => ctx.llm.registerModelDiscovery('llm-example', discover)).toThrow(/already registered/)
|
||||
expect(ctx.llm.listModelDiscoveryNamespaces()).toEqual(['llm-example'])
|
||||
})
|
||||
|
||||
it('normalizes what an interrogation returns without inventing capacities', async () => {
|
||||
const ctx = await setup()
|
||||
ctx.llm.registerModelDiscovery('llm-example', () => Promise.resolve([
|
||||
{ id: 'keep', name: 'Keep', contextWindow: 1024, maxTokens: 256 },
|
||||
{ id: '' },
|
||||
{ id: 'keep' },
|
||||
{ id: 'bare' },
|
||||
] as never))
|
||||
|
||||
expect(await ctx.llm.discoverModels('llm-example', { baseURL: 'https://gateway.example/v1' })).toEqual([
|
||||
{ id: 'keep', name: 'Keep', contextWindow: 1024, maxTokens: 256 },
|
||||
{ id: 'bare' },
|
||||
])
|
||||
})
|
||||
|
||||
it('refuses a namespace nothing serves and a draft with no endpoint', async () => {
|
||||
const ctx = await setup()
|
||||
ctx.llm.registerModelDiscovery('llm-example', () => Promise.resolve([]))
|
||||
|
||||
await expect(ctx.llm.discoverModels('llm-absent', { baseURL: 'https://gateway.example/v1' }))
|
||||
.rejects.toMatchObject({ code: 'NO_DISCOVERY' })
|
||||
await expect(ctx.llm.discoverModels('llm-example', { baseURL: '' }))
|
||||
.rejects.toMatchObject({ code: 'INVALID_DISCOVERY' })
|
||||
})
|
||||
})
|
||||
|
||||
Reference in New Issue
Block a user