Merge remote-tracking branch 'origin/master' into worktree/routed-model-compaction-policy
# Conflicts: # .agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml # .agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md # .agents/notes/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md # docs/cordis-catalog/events.md # docs/cordis-catalog/services.md # docs/event-producer-consumer.md # examples/headless-agent/tests/harness.ts # examples/repl-agent/cordis.yml # packages/compact/compact-basic/README.md # packages/compact/compact-basic/src/index.ts # packages/compact/compact-basic/tests/compact-basic.spec.ts # packages/compact/compact-basic/tests/loader-composition.spec.ts # packages/cordis/tool-cordis/src/api-catalog.ts # packages/llm/README.md # packages/llm/llm-deepseek/src/adapter.ts # packages/llm/llm-pi-ai/src/adapter.ts # packages/llm/llm/README.md # packages/llm/llm/src/index.ts # scripts/gen-cordis-catalog.ts # website/zh-CN/api/harness/events.md # website/zh-CN/api/harness/llm.md # website/zh-CN/api/harness/token-meter.md # website/zh-CN/guide/config.md
This commit is contained in:
@@ -16,7 +16,9 @@ import type {
|
||||
} from '@earendil-works/pi-ai'
|
||||
import { attributionHeaders, LlmAdapter, LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import type { GenerateOptions, LlmModelContext, LlmModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm'
|
||||
import type { PiAiProviderProfile } from './config.ts'
|
||||
import { idleWatchdog, timeoutOf } from '@deepseek-ai/dsh-timeout'
|
||||
import { resolveProfiles } from './config.ts'
|
||||
import type { PiAiProviderProfile, ResolvedPiAiProviderProfile } from './config.ts'
|
||||
import { toPiContext } from './context.ts'
|
||||
import { toStreamChunks } from './stream.ts'
|
||||
|
||||
@@ -48,8 +50,8 @@ function profileOptions(profile: PiAiProviderProfile): SimpleStreamOptions {
|
||||
...profile.transport === undefined ? {} : { transport: profile.transport },
|
||||
...profile.timeoutMs === undefined ? {} : { timeoutMs: profile.timeoutMs },
|
||||
...profile.websocketConnectTimeoutMs === undefined ? {} : { websocketConnectTimeoutMs: profile.websocketConnectTimeoutMs },
|
||||
...profile.maxRetries === undefined ? {} : { maxRetries: profile.maxRetries },
|
||||
...profile.maxRetryDelayMs === undefined ? {} : { maxRetryDelayMs: profile.maxRetryDelayMs },
|
||||
// The agent recovery layer owns visible attempts; one adapter call is one SDK attempt.
|
||||
maxRetries: 0,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -68,11 +70,11 @@ function requestHeaders(headers: Readonly<Record<string, string>> | undefined):
|
||||
* request, so models need not be registered during the Cordis lifecycle.
|
||||
*/
|
||||
export class PiAiAdapter extends LlmAdapter {
|
||||
private readonly profiles: ReadonlyMap<string, PiAiProviderProfile>
|
||||
private readonly profiles: ReadonlyMap<string, ResolvedPiAiProviderProfile>
|
||||
|
||||
constructor(options: PiAiAdapterOptions) {
|
||||
super()
|
||||
this.profiles = new Map(options.profiles.map(profile => [profile.provider, profile]))
|
||||
this.profiles = new Map(resolveProfiles(options.profiles).map(profile => [profile.provider, profile]))
|
||||
}
|
||||
|
||||
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
|
||||
@@ -113,12 +115,12 @@ export class PiAiAdapter extends LlmAdapter {
|
||||
}
|
||||
const model = resolveModel(profile, options.model)
|
||||
|
||||
// Pi-ai has no iterator-return cancellation hook. Chain an internal signal
|
||||
// and abort it when this generator exits so early consumers stop the HTTP stream.
|
||||
const controller = new AbortController()
|
||||
const onCallerAbort = (): void => { controller.abort(options.signal?.reason) }
|
||||
if (options.signal?.aborted) controller.abort(options.signal.reason)
|
||||
else options.signal?.addEventListener('abort', onCallerAbort, { once: true })
|
||||
const consumer = new AbortController()
|
||||
const upstream = options.signal === undefined
|
||||
? consumer.signal
|
||||
: AbortSignal.any([options.signal, consumer.signal])
|
||||
const streamIdleTimeoutMs = profile.streamIdleTimeoutMs
|
||||
using watchdog = idleWatchdog(upstream, streamIdleTimeoutMs, 'LLM_STREAM_IDLE_TIMEOUT')
|
||||
|
||||
try {
|
||||
const events = streamSimple(model, toPiContext(options), {
|
||||
@@ -126,15 +128,44 @@ export class PiAiAdapter extends LlmAdapter {
|
||||
...options.temperature === undefined ? {} : { temperature: options.temperature },
|
||||
...options.maxTokens === undefined ? {} : { maxTokens: options.maxTokens },
|
||||
...options.sessionId === undefined ? {} : { sessionId: String(options.sessionId) },
|
||||
signal: controller.signal,
|
||||
signal: watchdog.signal,
|
||||
// Profile headers are deployment-owned; attribution names are
|
||||
// Harness-owned and therefore win collisions.
|
||||
headers: requestHeaders(profile.headers),
|
||||
})
|
||||
yield* toStreamChunks(events, model.contextWindow)
|
||||
const iterator = toStreamChunks(events, model.contextWindow)[Symbol.asyncIterator]()
|
||||
let exhausted = false
|
||||
try {
|
||||
while (true) {
|
||||
const result = await watchdog.next(iterator)
|
||||
const timeout = timeoutOf(watchdog.signal, 'LLM_STREAM_IDLE_TIMEOUT')
|
||||
if (timeout !== undefined) throw timeout
|
||||
if (result.done) {
|
||||
exhausted = true
|
||||
return
|
||||
}
|
||||
yield result.value
|
||||
}
|
||||
} finally {
|
||||
if (!exhausted) {
|
||||
consumer.abort('pi-ai stream consumer stopped')
|
||||
try {
|
||||
await iterator.return(undefined)
|
||||
} catch (_abortedSdkTeardown) {
|
||||
// The stable signal already owns SDK termination; return-time abort cannot add an outcome.
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (error: unknown) {
|
||||
if (timeoutOf(watchdog.signal, 'LLM_STREAM_IDLE_TIMEOUT') !== undefined) {
|
||||
throw new LlmError(`pi-ai stream idle timeout after ${streamIdleTimeoutMs}ms`, 'TIMEOUT', { cause: error })
|
||||
}
|
||||
if (options.signal?.aborted) {
|
||||
throw new LlmError('pi-ai request aborted by caller', 'ABORTED', { cause: error })
|
||||
}
|
||||
throw error
|
||||
} finally {
|
||||
options.signal?.removeEventListener('abort', onCallerAbort)
|
||||
controller.abort('consumer stopped streaming')
|
||||
consumer.abort('pi-ai stream consumer stopped')
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,6 +7,10 @@
|
||||
import { getProviders } from '@earendil-works/pi-ai'
|
||||
import type { CacheRetention, ThinkingBudgets, ThinkingLevel, Transport } from '@earendil-works/pi-ai'
|
||||
import z from 'schemastery'
|
||||
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
|
||||
|
||||
/** Default maximum idle interval while an adapter stream read is outstanding. */
|
||||
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000
|
||||
|
||||
/** Configuration for one pi-ai provider route. */
|
||||
export interface PiAiProviderProfile {
|
||||
@@ -30,10 +34,14 @@ export interface PiAiProviderProfile {
|
||||
timeoutMs?: number
|
||||
/** WebSocket connection timeout in milliseconds. */
|
||||
websocketConnectTimeoutMs?: number
|
||||
/** Provider SDK retry count. */
|
||||
maxRetries?: number
|
||||
/** Maximum provider-requested retry delay in milliseconds. */
|
||||
maxRetryDelayMs?: number
|
||||
/** Maximum provider idle time while one stream read is outstanding. */
|
||||
streamIdleTimeoutMs?: number
|
||||
}
|
||||
|
||||
/** Validated profile with every adapter-owned default resolved. */
|
||||
export interface ResolvedPiAiProviderProfile extends PiAiProviderProfile {
|
||||
/** Positive finite provider-idle interval after defaulting. */
|
||||
streamIdleTimeoutMs: number
|
||||
}
|
||||
|
||||
/** Plugin configuration: the non-empty provider profiles this instance owns. */
|
||||
@@ -60,8 +68,7 @@ const profile = z.object({
|
||||
transport: z.union(['sse', 'websocket', 'websocket-cached', 'auto']),
|
||||
timeoutMs: z.natural(),
|
||||
websocketConnectTimeoutMs: z.natural(),
|
||||
maxRetries: z.natural(),
|
||||
maxRetryDelayMs: z.natural(),
|
||||
streamIdleTimeoutMs: z.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
|
||||
})
|
||||
|
||||
/** Runtime schema for {@link Config}. */
|
||||
@@ -75,11 +82,18 @@ export const Config: z<Config> = z.object({
|
||||
* @param profiles - configured provider profiles.
|
||||
* @returns validated profiles in configuration order.
|
||||
*/
|
||||
export function resolveProfiles(profiles: readonly PiAiProviderProfile[]): PiAiProviderProfile[] {
|
||||
export function resolveProfiles(profiles: readonly PiAiProviderProfile[]): ResolvedPiAiProviderProfile[] {
|
||||
if (profiles.length === 0) throw new Error('llm-pi-ai: providers must contain at least one profile')
|
||||
const supported = new Set<string>(getProviders())
|
||||
const seen = new Set<string>()
|
||||
return profiles.map((source) => {
|
||||
const legacy = source as PiAiProviderProfile & {
|
||||
maxRetries?: unknown
|
||||
maxRetryDelayMs?: unknown
|
||||
}
|
||||
if ('maxRetries' in legacy || 'maxRetryDelayMs' in legacy) {
|
||||
throw new Error('llm-pi-ai: maxRetries and maxRetryDelayMs were removed; compose agent recovery with dsh-llm-retry')
|
||||
}
|
||||
if (source.provider.length === 0) throw new Error('llm-pi-ai: provider names must be non-empty')
|
||||
if (!supported.has(source.provider)) throw new Error(`llm-pi-ai: unknown pi-ai provider "${source.provider}"`)
|
||||
if (seen.has(source.provider)) throw new Error(`llm-pi-ai: duplicate provider profile "${source.provider}"`)
|
||||
@@ -89,9 +103,18 @@ export function resolveProfiles(profiles: readonly PiAiProviderProfile[]): PiAiP
|
||||
if (source.baseURL !== undefined && source.baseURL.length === 0) {
|
||||
throw new Error(`llm-pi-ai: provider "${source.provider}" has an empty baseURL`)
|
||||
}
|
||||
const streamIdleTimeoutMs = source.streamIdleTimeoutMs ?? DEFAULT_STREAM_IDLE_TIMEOUT_MS
|
||||
if (!Number.isFinite(streamIdleTimeoutMs)
|
||||
|| streamIdleTimeoutMs <= 0
|
||||
|| streamIdleTimeoutMs > MAX_TIMER_DELAY_MS) {
|
||||
throw new Error(
|
||||
`llm-pi-ai: provider "${source.provider}" streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`,
|
||||
)
|
||||
}
|
||||
seen.add(source.provider)
|
||||
return {
|
||||
...source,
|
||||
streamIdleTimeoutMs,
|
||||
...source.headers === undefined ? {} : { headers: { ...source.headers } },
|
||||
...source.thinkingBudgets === undefined ? {} : { thinkingBudgets: { ...source.thinkingBudgets } },
|
||||
}
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
* @module dsh-llm-pi-ai/stream
|
||||
*/
|
||||
|
||||
import { CallId, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import { CallId, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmError, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm'
|
||||
import type { FinishReason, StreamChunk, TokenUsage } from '@deepseek-ai/dsh-llm'
|
||||
import { isContextOverflow } from '@earendil-works/pi-ai'
|
||||
import type { AssistantMessage, AssistantMessageEvent, Usage as PiUsage } from '@earendil-works/pi-ai'
|
||||
@@ -30,9 +30,15 @@ export function mapUsage(usage: PiUsage): TokenUsage {
|
||||
|
||||
function classifyPiAiError(message: string): string {
|
||||
if (/\b(?:401|403)\b/.test(message)) return 'AUTH'
|
||||
if (isQuotaExceededError(message)) return QUOTA_EXCEEDED_CODE
|
||||
if (/\b429\b|rate.?limit/i.test(message)) return 'RATE_LIMIT'
|
||||
if (/\b400\b|invalid.?request/i.test(message)) return 'INVALID_REQUEST'
|
||||
if (/\b5\d\d\b/.test(message)) return 'SERVER'
|
||||
if (/\btime(?:d)?\s*out\b|timeout/i.test(message)) return 'TIMEOUT'
|
||||
if (/\b(?:network|connection|socket|fetch)\b|\bECONN[A-Z]+\b/i.test(message)
|
||||
|| /\b(?:other side closed|HTTP2 request did not get a response|WebSocket closed unexpectedly)\b/i.test(message)) {
|
||||
return 'TRANSPORT'
|
||||
}
|
||||
return 'PI_AI_ERROR'
|
||||
}
|
||||
|
||||
@@ -52,8 +58,10 @@ export function mapStopReason(message: AssistantMessage, contextWindow?: number)
|
||||
if (piAiOverflow || harnessOverflow) {
|
||||
return {
|
||||
kind: 'error',
|
||||
message: message.errorMessage ?? `pi-ai detected context overflow for model "${message.model}"`,
|
||||
code: CONTEXT_WINDOW_EXCEEDED_CODE,
|
||||
failure: {
|
||||
message: message.errorMessage ?? `pi-ai detected context overflow for model "${message.model}"`,
|
||||
code: CONTEXT_WINDOW_EXCEEDED_CODE,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -61,10 +69,13 @@ export function mapStopReason(message: AssistantMessage, contextWindow?: number)
|
||||
case 'stop': return { kind: 'stop' }
|
||||
case 'length': return { kind: 'max-tokens' }
|
||||
case 'toolUse': return { kind: 'tool-calls' }
|
||||
case 'aborted': return { kind: 'aborted' }
|
||||
case 'aborted': return {
|
||||
kind: 'aborted',
|
||||
failure: { message: message.errorMessage ?? 'pi-ai stream aborted', code: 'ABORTED' },
|
||||
}
|
||||
case 'error': {
|
||||
const text = message.errorMessage ?? 'pi-ai stream error'
|
||||
return { kind: 'error', message: text, code: classifyPiAiError(text) }
|
||||
return { kind: 'error', failure: { message: text, code: classifyPiAiError(text) } }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user