feat(llm): interrogate a draft provider endpoint for its models
Once a pi-ai route became a declaration rather than a catalog lookup, adding an OpenAI-compatible gateway meant knowing its model ids up front. Most such endpoints publish that list at `GET /models`, but no seam operation could ask: every one is keyed by a registered provider route, and the provider being added has no route, no stored profile, and no stored credential — the endpoint and key are values in a form. Interrogation is therefore keyed by settings namespace, which a configuration surface already holds from the configurable-provider directory. `registerModelDiscovery` offers it per namespace, `discoverModels` asks, and the request carries the draft itself. The reply is candidates, not a catalog: every field but the id is optional because most listings disclose nothing else, and adopting one is a settings write like any other. Nothing here reads or writes settings or credentials, so `settings.yaml` still decides what a route serves. `llm.discoverModels` carries the same draft over the wire. Its apiKey is the third and last payload a secret may ride, and it is never stored, logged, or echoed; every refusal folds into `model-discovery-failed`, naming the endpoint asked but never the credential offered. The pi-ai side is a plain GET for OpenAI-compatible protocols only — their listing shape is the one gateways, self-hosted servers, and the official endpoints agree on. Others say so, sending the user to hand-entry rather than reporting a guessed shape as an empty provider. The reply is read under a four-megabyte ceiling held on the bytes actually received, because the endpoint is a URL the user typed.
This commit is contained in:
207
packages/llm/llm-pi-ai/src/discovery.ts
Normal file
207
packages/llm/llm-pi-ai/src/discovery.ts
Normal file
@@ -0,0 +1,207 @@
|
||||
/**
|
||||
* One-shot interrogation of a provider endpoint's model listing, serving the
|
||||
* configuration surface's "fetch available models" action.
|
||||
*
|
||||
* This is deliberately *not* a catalog refresh. Nothing here is stored: the
|
||||
* request carries a draft the user is still editing — an endpoint and a
|
||||
* credential neither of which may exist in `settings.yaml` yet — and the reply
|
||||
* is candidate metadata the surface offers for adoption. `settings.yaml`
|
||||
* remains the only thing that decides what a route serves.
|
||||
*
|
||||
* Only OpenAI-compatible protocols are interrogated. Their listing is the one
|
||||
* shape a gateway, a self-hosted server, and the official endpoints all agree
|
||||
* on, which is the case this action exists for; every other protocol reports
|
||||
* that it cannot be interrogated so the surface falls back to hand-entry
|
||||
* rather than guessing a response shape.
|
||||
*
|
||||
* @module dsh-llm-pi-ai/discovery
|
||||
*/
|
||||
|
||||
import { LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import type { LlmDiscoveredModel, LlmModelDiscoveryRequest } from '@deepseek-ai/dsh-llm'
|
||||
import { attributionHeaders } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
/**
|
||||
* Protocols whose model listing this module can read. Every entry speaks
|
||||
* OpenAI's `GET /models` shape; pi-ai's other protocols are absent because a
|
||||
* wrong guess at their response shape would be reported as an empty provider
|
||||
* rather than as the gap it is.
|
||||
*/
|
||||
const LISTABLE_PROTOCOLS: ReadonlySet<string> = new Set([
|
||||
'azure-openai-responses',
|
||||
'openai-codex-responses',
|
||||
'openai-completions',
|
||||
'openai-responses',
|
||||
])
|
||||
|
||||
/**
|
||||
* Endpoint replies larger than this are refused. The endpoint is whatever URL
|
||||
* the user typed, so the ceiling holds on the bytes actually read rather than
|
||||
* on the length the server claims — the same two-stage shape `dsh-web-fetch`
|
||||
* uses for its own caller-supplied URLs, except that a truncated model listing
|
||||
* is not parseable, so overflow rejects instead of truncating.
|
||||
*/
|
||||
const MAX_RESPONSE_BYTES = 4 * 1024 * 1024
|
||||
|
||||
/** One entry of an OpenAI-compatible `GET /models` reply. */
|
||||
interface ListingEntry {
|
||||
id?: unknown
|
||||
/** Common gateway extensions; absent from the official listings. */
|
||||
name?: unknown
|
||||
display_name?: unknown
|
||||
context_window?: unknown
|
||||
context_length?: unknown
|
||||
max_tokens?: unknown
|
||||
max_output_tokens?: unknown
|
||||
}
|
||||
|
||||
/** A positive integer field of a listing entry, or `undefined` when absent or unusable. */
|
||||
function capacity(...candidates: readonly unknown[]): number | undefined {
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === 'number' && Number.isInteger(candidate) && candidate > 0) return candidate
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/** A non-empty string field of a listing entry, or `undefined`. */
|
||||
function label(...candidates: readonly unknown[]): string | undefined {
|
||||
for (const candidate of candidates) {
|
||||
if (typeof candidate === 'string' && candidate.length > 0) return candidate
|
||||
}
|
||||
return undefined
|
||||
}
|
||||
|
||||
/**
|
||||
* Join the endpoint base with the listing path. The base is treated as a
|
||||
* prefix rather than a URL to resolve against, so a deployment path such as
|
||||
* `https://gateway.example/openai/v1` keeps its segments instead of losing
|
||||
* them to `URL` resolution.
|
||||
*/
|
||||
function listingUrl(baseURL: string): string {
|
||||
return `${baseURL.replace(/\/+$/, '')}/models`
|
||||
}
|
||||
|
||||
/**
|
||||
* Read a reply body, refusing one that outgrows the ceiling. A declared length
|
||||
* is checked first so an honest server is turned away without transferring
|
||||
* anything; the accumulated total is what actually enforces the bound, because
|
||||
* a server that under-declares (or streams) tells us nothing up front.
|
||||
*/
|
||||
async function readBounded(response: Response, url: string): Promise<string> {
|
||||
const oversized = (): LlmError =>
|
||||
new LlmError(`${url} answered with more than ${MAX_RESPONSE_BYTES} bytes`, 'DISCOVERY_FAILED')
|
||||
const declared = Number(response.headers.get('content-length') ?? Number.NaN)
|
||||
if (Number.isFinite(declared) && declared > MAX_RESPONSE_BYTES) {
|
||||
await response.body?.cancel()
|
||||
throw oversized()
|
||||
}
|
||||
/* v8 ignore next -- fetch always exposes a body stream on a 2xx Response; the null guard is defensive. */
|
||||
if (response.body === null) return ''
|
||||
const reader = response.body.getReader()
|
||||
const chunks: Uint8Array[] = []
|
||||
let total = 0
|
||||
try {
|
||||
for (;;) {
|
||||
const { done, value } = await reader.read()
|
||||
if (done) break
|
||||
total += value.byteLength
|
||||
if (total > MAX_RESPONSE_BYTES) throw oversized()
|
||||
chunks.push(value)
|
||||
}
|
||||
} finally {
|
||||
/* v8 ignore next 4 -- cancel() after a completed or abandoned read settles without rejecting; unobserved best-effort cleanup. */
|
||||
await reader.cancel().catch(() => {
|
||||
// Cancel after a drained read, or after this function walked away from
|
||||
// an oversized one, is cleanup; the reply is already decided either way.
|
||||
})
|
||||
}
|
||||
const body = new Uint8Array(total)
|
||||
let offset = 0
|
||||
for (const chunk of chunks) {
|
||||
body.set(chunk, offset)
|
||||
offset += chunk.byteLength
|
||||
}
|
||||
return new TextDecoder().decode(body)
|
||||
}
|
||||
|
||||
/**
|
||||
* Read one OpenAI-compatible listing reply. Entries without a usable id are
|
||||
* skipped rather than failing the whole interrogation: a single malformed row
|
||||
* should not deny the user the rest of a working endpoint's catalog.
|
||||
*/
|
||||
function readListing(body: unknown): LlmDiscoveredModel[] {
|
||||
const data = (body as { data?: unknown } | null)?.data
|
||||
if (!Array.isArray(data)) {
|
||||
throw new LlmError(
|
||||
'the endpoint\'s model listing has no "data" array; enter this provider\'s models by hand',
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
const models: LlmDiscoveredModel[] = []
|
||||
for (const raw of data) {
|
||||
const entry = raw as ListingEntry | null
|
||||
const id = label(entry?.id)
|
||||
if (id === undefined) continue
|
||||
const name = label(entry?.name, entry?.display_name)
|
||||
const contextWindow = capacity(entry?.context_window, entry?.context_length)
|
||||
const maxTokens = capacity(entry?.max_output_tokens, entry?.max_tokens)
|
||||
models.push({
|
||||
id,
|
||||
...name === undefined ? {} : { name },
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
...maxTokens === undefined ? {} : { maxTokens },
|
||||
})
|
||||
}
|
||||
return models
|
||||
}
|
||||
|
||||
/**
|
||||
* Interrogate one draft provider endpoint for the models it advertises.
|
||||
* @param request - the endpoint, protocol, and one-shot credential to use.
|
||||
* @returns the advertised models in endpoint order.
|
||||
* @throws LlmError when the protocol has no readable listing, the endpoint
|
||||
* refuses or fails the request, or the reply is not a model listing.
|
||||
*/
|
||||
export async function discoverModels(
|
||||
request: LlmModelDiscoveryRequest,
|
||||
): Promise<readonly LlmDiscoveredModel[]> {
|
||||
const api = request.api ?? 'openai-completions'
|
||||
if (!LISTABLE_PROTOCOLS.has(api)) {
|
||||
throw new LlmError(
|
||||
`pi-ai protocol "${api}" has no model listing this build can read; enter this provider's models by hand`,
|
||||
'DISCOVERY_UNSUPPORTED',
|
||||
)
|
||||
}
|
||||
const url = listingUrl(request.baseURL)
|
||||
let response: Response
|
||||
try {
|
||||
response = await fetch(url, {
|
||||
method: 'GET',
|
||||
headers: {
|
||||
accept: 'application/json',
|
||||
...request.apiKey === undefined ? {} : { authorization: `Bearer ${request.apiKey}` },
|
||||
...attributionHeaders(),
|
||||
},
|
||||
...request.signal === undefined ? {} : { signal: request.signal },
|
||||
})
|
||||
} catch (error: unknown) {
|
||||
if (request.signal?.aborted) {
|
||||
throw new LlmError('model discovery aborted by caller', 'ABORTED', { cause: error })
|
||||
}
|
||||
throw new LlmError(`could not reach ${url}`, 'DISCOVERY_FAILED', { cause: error })
|
||||
}
|
||||
if (!response.ok) {
|
||||
throw new LlmError(
|
||||
`${url} answered ${response.status}${response.status === 401 || response.status === 403 ? '; check the API key' : ''}`,
|
||||
'DISCOVERY_FAILED',
|
||||
)
|
||||
}
|
||||
const text = await readBounded(response, url)
|
||||
let body: unknown
|
||||
try {
|
||||
body = JSON.parse(text)
|
||||
} catch (error: unknown) {
|
||||
throw new LlmError(`${url} did not answer with JSON`, 'DISCOVERY_FAILED', { cause: error })
|
||||
}
|
||||
return readListing(body)
|
||||
}
|
||||
@@ -50,6 +50,7 @@ import { PiAiAdapter } from './adapter.ts'
|
||||
import { catalogProviderIds } from './catalog.ts'
|
||||
import { assertServiceable, Config, resolveProfiles } from './config.ts'
|
||||
import type { ResolvedPiAiProviderProfile } from './config.ts'
|
||||
import { discoverModels } from './discovery.ts'
|
||||
|
||||
export { PiAiAdapter } from './adapter.ts'
|
||||
export type { PiAiAdapterOptions } from './adapter.ts'
|
||||
@@ -176,6 +177,10 @@ export function apply(ctx: Context, config: Config): void {
|
||||
directoryFacts = entries
|
||||
}
|
||||
ensureDirectory()
|
||||
// Interrogating an endpoint is a configuration-time action over a draft, so
|
||||
// it is offered for the whole namespace rather than per route: the provider
|
||||
// a surface is adding does not exist yet.
|
||||
ctx.llm.registerModelDiscovery(NS, discoverModels)
|
||||
// Route effects bind to this apply fiber via the stable `ctx` reference,
|
||||
// even when a swap runs inside the scoped settings callback below. A bare
|
||||
// mount (zero routes) is the dormant posture: nothing registers until a
|
||||
|
||||
Reference in New Issue
Block a user