Merge remote-tracking branch 'origin/master' into worktree/pr628-merge-20260727
# Conflicts: # .agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md # docs/architecture.i18n.yaml # docs/config-catalog.md # docs/cordis-catalog/events.md # docs/cordis-catalog/services.md # docs/core-data-structures/llm-streaming.i18n.yaml # docs/core-data-structures/llm-streaming.md # docs/core-data-structures/llm-streaming.zh.md # docs/event-producer-consumer.md # docs/module-graph.md # examples/headless-agent/tests/headless.snapshot.ts # packages/compact/compact-basic/tests/compact-loop-repro.spec.ts # packages/cordis/tool-cordis/src/api-catalog.ts # packages/core/agent-loop/README.md # packages/core/agent-loop/src/loop.ts # packages/examples/agent-spine-demo/README.md # packages/llm/README.md # packages/llm/llm-deepseek/src/adapter.ts # packages/llm/llm-pi-ai/src/adapter.ts # packages/llm/llm-retry/README.md # packages/llm/llm/README.md # packages/llm/llm/src/index.ts # packages/llm/llm/tests/service.spec.ts # packages/support/llm-replay/src/index.ts # packages/support/llm-replay/tests/llm-replay.spec.ts
This commit is contained in:
@@ -5,12 +5,12 @@
|
||||
* @module dsh-llm-deepseek/adapter
|
||||
*/
|
||||
|
||||
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
|
||||
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
|
||||
import type {
|
||||
GenerateOptions,
|
||||
LlmModelContext,
|
||||
LlmModelInfo,
|
||||
LlmProviderInfo,
|
||||
LlmResolvedModelInfo,
|
||||
ResolvedRetryPolicy,
|
||||
RetryPolicyConfig,
|
||||
StreamChunk,
|
||||
@@ -22,7 +22,7 @@ import { parseSse } from './sse.ts'
|
||||
import { translate } from './translate.ts'
|
||||
import type { WireError } from './types.ts'
|
||||
|
||||
/** One optional model entry advertised by the hand-written adapter. */
|
||||
/** One optional model entry advertised by the direct-fetch adapter. */
|
||||
export interface DeepSeekCatalogModel {
|
||||
/** Wire model id accepted by the configured endpoint. */
|
||||
id: string
|
||||
@@ -55,6 +55,26 @@ export interface DeepSeekAdapterOptions {
|
||||
/** Default maximum idle interval while an adapter stream read is outstanding. */
|
||||
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000
|
||||
const STREAM_IDLE_TIMEOUT_CODE = 'LLM_STREAM_IDLE_TIMEOUT'
|
||||
const OFF_REASONING_EFFORT = ReasoningEffortId('off')
|
||||
const HIGH_REASONING_EFFORT = ReasoningEffortId('high')
|
||||
const MAX_REASONING_EFFORT = ReasoningEffortId('max')
|
||||
const REASONING_EFFORTS = [
|
||||
{ id: OFF_REASONING_EFFORT, name: 'Off' },
|
||||
{ id: HIGH_REASONING_EFFORT, name: 'High' },
|
||||
{ id: MAX_REASONING_EFFORT, name: 'Max' },
|
||||
] as const
|
||||
const OFF_ONLY_REASONING_EFFORTS = [
|
||||
{ id: OFF_REASONING_EFFORT, name: 'Off' },
|
||||
] as const
|
||||
|
||||
function modelInfo(provider: string, model: DeepSeekCatalogModel): LlmModelInfo {
|
||||
return {
|
||||
provider,
|
||||
id: model.id,
|
||||
name: model.name ?? model.id,
|
||||
...model.description === undefined ? {} : { description: model.description },
|
||||
}
|
||||
}
|
||||
|
||||
function providerRetryAfterMs(value: string | null): number | undefined {
|
||||
if (value === null) return undefined
|
||||
@@ -103,6 +123,11 @@ export class DeepSeekAdapter extends LlmAdapter {
|
||||
|
||||
constructor(private readonly options: DeepSeekAdapterOptions) {
|
||||
super()
|
||||
if (options.defaults?.thinking === 'disabled'
|
||||
&& options.defaults.reasoningEffort !== undefined
|
||||
&& options.defaults.reasoningEffort !== 'off') {
|
||||
throw new Error('llm-deepseek: only reasoningEffort "off" can be configured when thinking is disabled')
|
||||
}
|
||||
if (options.defaultContextWindow !== undefined
|
||||
&& (!Number.isInteger(options.defaultContextWindow) || options.defaultContextWindow <= 0)) {
|
||||
throw new Error('llm-deepseek: defaultContextWindow must be a positive integer')
|
||||
@@ -127,21 +152,40 @@ export class DeepSeekAdapter extends LlmAdapter {
|
||||
}
|
||||
|
||||
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
|
||||
return Promise.resolve((this.options.models ?? []).map(model => ({
|
||||
provider,
|
||||
id: model.id,
|
||||
name: model.name ?? model.id,
|
||||
...model.description === undefined ? {} : { description: model.description },
|
||||
})))
|
||||
return Promise.resolve((this.options.models ?? []).map(model => modelInfo(provider, model)))
|
||||
}
|
||||
|
||||
override resolveModelContext(
|
||||
_provider: string,
|
||||
override resolveModel(
|
||||
provider: string,
|
||||
model: string,
|
||||
): Promise<LlmModelContext | undefined> {
|
||||
const contextWindow = this.options.models?.find(entry => entry.id === model)?.contextWindow
|
||||
_signal?: AbortSignal,
|
||||
): Promise<LlmResolvedModelInfo> {
|
||||
const configured = this.options.models?.find(entry => entry.id === model)
|
||||
const contextWindow = configured?.contextWindow
|
||||
?? this.options.defaultContextWindow
|
||||
return Promise.resolve(contextWindow === undefined ? undefined : { contextWindow })
|
||||
return Promise.resolve({
|
||||
...configured === undefined
|
||||
? { provider, id: model, name: model }
|
||||
: modelInfo(provider, configured),
|
||||
...contextWindow === undefined ? {} : { context: { contextWindow } },
|
||||
...this.options.defaults?.thinking === 'disabled'
|
||||
? {
|
||||
reasoning: {
|
||||
efforts: OFF_ONLY_REASONING_EFFORTS,
|
||||
defaultEffort: OFF_REASONING_EFFORT,
|
||||
},
|
||||
}
|
||||
: {
|
||||
reasoning: {
|
||||
efforts: REASONING_EFFORTS,
|
||||
defaultEffort: this.options.defaults?.reasoningEffort === 'off'
|
||||
? OFF_REASONING_EFFORT
|
||||
: this.options.defaults?.reasoningEffort === 'max'
|
||||
? MAX_REASONING_EFFORT
|
||||
: HIGH_REASONING_EFFORT,
|
||||
},
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
async * stream(options: GenerateOptions): AsyncIterable<StreamChunk> {
|
||||
|
||||
@@ -29,18 +29,19 @@ const DEFAULT_MODELS: DeepSeekCatalogModel[] = [
|
||||
/**
|
||||
* Plugin config, validated by the same-named schemastery schema. Every field
|
||||
* is optional in yml: credentials/endpoint fall back to the environment (a
|
||||
* missing API key fails plugin load, not the first call), and omitted
|
||||
* thinking fields send nothing on the wire, so the provider default applies.
|
||||
* missing API key fails plugin load, not the first call), omitted thinking
|
||||
* mode uses the provider default, and omitted reasoning effort resolves to
|
||||
* `high`.
|
||||
*/
|
||||
export interface Config {
|
||||
/** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */
|
||||
apiKey?: string
|
||||
/** Endpoint base; falls back to $DEEPSEEK_BASE_URL, then the public API. */
|
||||
baseURL?: string
|
||||
/** Thinking-mode default for every request (provider default: enabled). */
|
||||
/** Deployment thinking policy; `disabled` limits every conversation request to `off`. */
|
||||
thinking?: 'enabled' | 'disabled'
|
||||
/** Thinking effort (only meaningful with thinking enabled). */
|
||||
reasoningEffort?: 'high' | 'max'
|
||||
/** Default thinking effort (default `high`); `off` disables thinking per request. */
|
||||
reasoningEffort?: 'off' | 'high' | 'max'
|
||||
/** Positive context capacity used when the selected model has no exact value. */
|
||||
defaultContextWindow?: number
|
||||
/** Advisory models shown by discovery consumers; defaults to V4 Flash and V4 Pro. */
|
||||
@@ -62,7 +63,7 @@ export const Config: z<Config> = z.object({
|
||||
apiKey: z.string(),
|
||||
baseURL: z.string(),
|
||||
thinking: z.union(['enabled', 'disabled']),
|
||||
reasoningEffort: z.union(['high', 'max']),
|
||||
reasoningEffort: z.union(['off', 'high', 'max']),
|
||||
defaultContextWindow: z.number().step(1).min(1),
|
||||
models: z.array(catalogModel).default(DEFAULT_MODELS),
|
||||
streamIdleTimeoutMs: z.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
|
||||
@@ -98,6 +99,11 @@ function resolveModels(models: readonly DeepSeekCatalogModel[] | undefined): Dee
|
||||
}
|
||||
|
||||
export function apply(ctx: Context, config: Config): void {
|
||||
if (config.thinking === 'disabled'
|
||||
&& config.reasoningEffort !== undefined
|
||||
&& config.reasoningEffort !== 'off') {
|
||||
throw new Error('llm-deepseek: only reasoningEffort "off" can be configured when thinking is disabled')
|
||||
}
|
||||
const apiKey = config.apiKey ?? process.env.DEEPSEEK_API_KEY
|
||||
if (apiKey === undefined || apiKey.length === 0) {
|
||||
throw new Error('llm-deepseek: an API key is required (Config.apiKey or $DEEPSEEK_API_KEY)')
|
||||
|
||||
@@ -6,13 +6,49 @@
|
||||
* @module dsh-llm-deepseek/serialize
|
||||
*/
|
||||
|
||||
import { LlmError } from '@deepseek-ai/dsh-llm'
|
||||
import type { ContentBlock, GenerateOptions, Message } from '@deepseek-ai/dsh-llm'
|
||||
import type { WireMessage, WireRequest, WireTool } from './types.ts'
|
||||
|
||||
/** Adapter-level request defaults (from plugin config). */
|
||||
export interface RequestDefaults {
|
||||
thinking?: 'enabled' | 'disabled' | undefined
|
||||
reasoningEffort?: 'high' | 'max' | undefined
|
||||
reasoningEffort?: 'off' | 'high' | 'max' | undefined
|
||||
}
|
||||
|
||||
interface ResolvedThinking {
|
||||
thinking?: 'enabled' | 'disabled'
|
||||
reasoningEffort?: 'high' | 'max'
|
||||
}
|
||||
|
||||
/** Validate the adapter-owned effort before resolving its DeepSeek wire fields. */
|
||||
function reasoningEffort(effort: NonNullable<GenerateOptions['reasoningEffort']>): 'off' | 'high' | 'max' {
|
||||
if (effort === 'off' || effort === 'high' || effort === 'max') {
|
||||
return effort as 'off' | 'high' | 'max'
|
||||
}
|
||||
throw new LlmError(
|
||||
`DeepSeek does not support reasoning effort "${effort}"`,
|
||||
'UNSUPPORTED_REASONING_EFFORT',
|
||||
)
|
||||
}
|
||||
|
||||
/** Resolve one legal thinking/effort pair without exposing `off` as a wire effort. */
|
||||
function resolveThinking(options: GenerateOptions, defaults: RequestDefaults): ResolvedThinking {
|
||||
if (options.purpose === 'session-title') return { thinking: 'disabled' }
|
||||
const effort = options.reasoningEffort === undefined
|
||||
? defaults.reasoningEffort
|
||||
: reasoningEffort(options.reasoningEffort)
|
||||
if (defaults.thinking === 'disabled' && effort !== undefined && effort !== 'off') {
|
||||
throw new LlmError(
|
||||
`DeepSeek deployment does not support reasoning effort "${effort}"`,
|
||||
'UNSUPPORTED_REASONING_EFFORT',
|
||||
)
|
||||
}
|
||||
if (effort === 'off') return { thinking: 'disabled' }
|
||||
if (effort === 'high' || effort === 'max') {
|
||||
return { thinking: 'enabled', reasoningEffort: effort }
|
||||
}
|
||||
return defaults.thinking === undefined ? {} : { thinking: defaults.thinking }
|
||||
}
|
||||
|
||||
/** Join the text blocks of a message (used for user/tool-result content). */
|
||||
@@ -120,16 +156,17 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa
|
||||
}))
|
||||
// A short title budget must produce visible text; conversation and
|
||||
// compaction calls continue to inherit the adapter's thinking defaults.
|
||||
const thinking = options.purpose === 'session-title' ? 'disabled' : defaults.thinking
|
||||
const reasoningEffort = options.purpose === 'session-title' ? undefined : defaults.reasoningEffort
|
||||
const resolvedThinking = resolveThinking(options, defaults)
|
||||
|
||||
return {
|
||||
model: options.model,
|
||||
messages,
|
||||
stream: true,
|
||||
stream_options: { include_usage: true },
|
||||
...thinking !== undefined ? { thinking: { type: thinking } } : {},
|
||||
...reasoningEffort !== undefined ? { reasoning_effort: reasoningEffort } : {},
|
||||
...resolvedThinking.thinking !== undefined ? { thinking: { type: resolvedThinking.thinking } } : {},
|
||||
...resolvedThinking.reasoningEffort !== undefined
|
||||
? { reasoning_effort: resolvedThinking.reasoningEffort }
|
||||
: {},
|
||||
...tools !== undefined && tools.length > 0 ? { tools } : {},
|
||||
...options.temperature !== undefined ? { temperature: options.temperature } : {},
|
||||
...options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {},
|
||||
|
||||
@@ -1,65 +1,33 @@
|
||||
/**
|
||||
* Decode an SSE byte stream into event `data` payloads. Network reads may split UTF-8 or lines;
|
||||
* CRLF, comments, non-data fields, and multi-data events are handled per SSE rules. The literal
|
||||
* `[DONE]` is yielded so the caller owns final flushing, and EOF before it raises {@link LlmError}.
|
||||
* Decode an SSE byte stream into event `data` payloads. Framing — chunk
|
||||
* reassembly, UTF-8/CRLF/BOM handling, comment and non-data field skipping,
|
||||
* multi-`data:` joining — is `eventsource-parser`'s; this module keeps only
|
||||
* the DeepSeek protocol: the literal `[DONE]` is yielded so the caller owns
|
||||
* final flushing, and EOF before it raises {@link LlmError}. Framing is
|
||||
* spec-strict: an event dispatches only on its blank-line terminator, so an
|
||||
* unterminated tail at EOF is truncation, not a flushable payload.
|
||||
*
|
||||
* Minimal SSE (text/event-stream) parser for the chat-completions stream.
|
||||
* @module dsh-llm-deepseek/sse
|
||||
*/
|
||||
|
||||
import { EventSourceParserStream } from 'eventsource-parser/stream'
|
||||
import { LlmError } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
/** The terminal payload DeepSeek (and OpenAI) send after the last chunk. */
|
||||
export const DONE = '[DONE]'
|
||||
|
||||
/** Extract the joined data payload from one raw SSE event block. */
|
||||
function eventData(block: string): string | undefined {
|
||||
const data: string[] = []
|
||||
for (const rawLine of block.split('\n')) {
|
||||
const line = rawLine.endsWith('\r') ? rawLine.slice(0, -1) : rawLine
|
||||
if (line.startsWith('data:')) {
|
||||
// The spec strips ONE leading space after the colon.
|
||||
data.push(line.startsWith('data: ') ? line.slice(6) : line.slice(5))
|
||||
}
|
||||
// Comments (':…') and other fields (event:, id:, retry:) are ignored.
|
||||
}
|
||||
if (data.length === 0) return undefined
|
||||
return data.join('\n')
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a byte stream into SSE data payloads. Yields `[DONE]` as the final
|
||||
* Parse an SSE byte stream into data payloads. Yields `[DONE]` as the final
|
||||
* value and returns; throws `LlmError('STREAM_CLOSED')` when the stream ends
|
||||
* without it (truncated response — the model call cannot be trusted).
|
||||
* @param stream - raw SSE bytes; reads may split anywhere, including mid-UTF-8 sequence.
|
||||
* @returns each event's data payload in arrival order, the `[DONE]` sentinel last.
|
||||
*/
|
||||
export async function* parseSse(stream: AsyncIterable<Uint8Array>): AsyncGenerator<string> {
|
||||
const decoder = new TextDecoder()
|
||||
let buffer = ''
|
||||
|
||||
for await (const bytes of stream) {
|
||||
buffer += decoder.decode(bytes, { stream: true })
|
||||
// Events are separated by a blank line (\n\n; tolerate \r\n\r\n via the
|
||||
// per-line \r strip in eventData and a normalized split here).
|
||||
let boundary: number
|
||||
while ((boundary = buffer.search(/\r?\n\r?\n/)) !== -1) {
|
||||
const matched = /\r?\n\r?\n/.exec(buffer.slice(boundary))
|
||||
const block = buffer.slice(0, boundary)
|
||||
// matched cannot be null: search() just found the same pattern at 0.
|
||||
buffer = buffer.slice(boundary + (matched as RegExpExecArray)[0].length)
|
||||
const data = eventData(block)
|
||||
if (data === undefined) continue
|
||||
yield data
|
||||
if (data === DONE) return
|
||||
}
|
||||
}
|
||||
|
||||
// Flush any final un-terminated event (servers usually end with \n\n, but
|
||||
// a trailing block without one is still parseable).
|
||||
buffer += decoder.decode()
|
||||
const data = eventData(buffer)
|
||||
if (data !== undefined) {
|
||||
export async function* parseSse(stream: ReadableStream<BufferSource>): AsyncGenerator<string> {
|
||||
const events = stream
|
||||
.pipeThrough(new TextDecoderStream())
|
||||
.pipeThrough(new EventSourceParserStream())
|
||||
for await (const { data } of events) {
|
||||
yield data
|
||||
if (data === DONE) return
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user