Merge remote-tracking branch 'origin/master' into worktree/web-multimodal-image-input

# Conflicts:
#	docs/architecture.i18n.yaml
#	docs/architecture.md
#	docs/architecture.zh.md
#	docs/config-catalog.md
#	docs/core-data-structures/core.i18n.yaml
#	docs/core-data-structures/llm-streaming.i18n.yaml
#	docs/module-graph.md
#	examples/acp-agent/tests/snapshots/cordis-inspect-jsdoc/session.jsonl
#	packages/README.i18n.yaml
#	packages/client/connection/src/client/fixture.ts
#	packages/client/connection/src/index.ts
#	packages/client/runtime/README.i18n.yaml
#	packages/client/runtime/README.md
#	packages/client/runtime/README.zh.md
#	packages/client/runtime/src/client/sessions/conversation.ts
#	packages/client/ui-conversation/README.i18n.yaml
#	packages/client/ui-conversation/src/client/apply.ts
#	packages/client/ui-conversation/src/client/chat/ChatView.tsx
#	packages/client/ui-conversation/src/client/chat/MessageItem.tsx
#	packages/client/ui-conversation/src/client/contract/slots.ts
#	packages/client/ui-trajectory/tests/views.spec.tsx
#	packages/compact/compact-basic/README.i18n.yaml
#	packages/cordis/tool-cordis/src/api-catalog.ts
#	packages/host/apiproxy/src/api-proxy.ts
#	packages/host/apiproxy/src/api/index.ts
#	packages/host/apiproxy/src/api/sessions.ts
#	packages/host/apiproxy/src/index.ts
#	packages/host/apiproxy/tests/fetch-carrier.spec.ts
#	packages/llm/llm-deepseek/src/adapter.ts
#	packages/llm/llm-deepseek/tests/adapter.spec.ts
#	packages/llm/llm-deepseek/tests/serialize.spec.ts
#	packages/llm/llm-pi-ai/README.i18n.yaml
#	packages/llm/llm-pi-ai/src/adapter.ts
#	packages/llm/llm-pi-ai/src/index.ts
#	packages/llm/llm-pi-ai/tests/adapter.spec.ts
#	packages/llm/llm/src/types.ts
#	packages/ui/tui/README.i18n.yaml
#	packages/ui/tui/src/index.ts
#	packages/ui/tui/tests/tui.spec.ts
This commit is contained in:
Yichen Jiang
2026-07-28 11:41:40 +08:00
1499 changed files with 48621 additions and 21956 deletions

View File

@@ -5,12 +5,14 @@
* @module dsh-llm-deepseek/adapter
*/
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm'
import { attributionHeaders, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
import type {
GenerateOptions,
LlmModelContext,
LlmModelInfo,
LlmProviderInfo,
LlmResolvedModelInfo,
ResolvedRetryPolicy,
RetryPolicyConfig,
StreamChunk,
} from '@deepseek-ai/dsh-llm'
import { idleWatchdog, MAX_TIMER_DELAY_MS, timeoutOf } from '@deepseek-ai/dsh-timeout'
@@ -20,7 +22,7 @@ import { parseSse } from './sse.ts'
import { translate } from './translate.ts'
import type { WireError } from './types.ts'
/** One optional model entry advertised by the hand-written adapter. */
/** One optional model entry advertised by the direct-fetch adapter. */
export interface DeepSeekCatalogModel {
/** Wire model id accepted by the configured endpoint. */
id: string
@@ -46,11 +48,35 @@ export interface DeepSeekAdapterOptions {
models?: readonly DeepSeekCatalogModel[]
/** Maximum provider idle time while one stream read is outstanding. */
streamIdleTimeoutMs?: number
/** Provider-owned model-request retry policy; omission uses normal defaults. */
retryPolicy?: RetryPolicyConfig
}
/** Default maximum idle interval while an adapter stream read is outstanding. */
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000
const STREAM_IDLE_TIMEOUT_CODE = 'LLM_STREAM_IDLE_TIMEOUT'
const OFF_REASONING_EFFORT = ReasoningEffortId('off')
const HIGH_REASONING_EFFORT = ReasoningEffortId('high')
const MAX_REASONING_EFFORT = ReasoningEffortId('max')
const REASONING_EFFORTS = [
{ id: OFF_REASONING_EFFORT, name: 'Off' },
{ id: HIGH_REASONING_EFFORT, name: 'High' },
{ id: MAX_REASONING_EFFORT, name: 'Max' },
] as const
const OFF_ONLY_REASONING_EFFORTS = [
{ id: OFF_REASONING_EFFORT, name: 'Off' },
] as const
function modelInfo(provider: string, model: DeepSeekCatalogModel): LlmModelInfo {
return {
provider,
id: model.id,
name: model.name ?? model.id,
...model.description === undefined ? {} : { description: model.description },
inputModalities: ['text'],
outputModalities: ['text'],
}
}
function providerRetryAfterMs(value: string | null): number | undefined {
if (value === null) return undefined
@@ -95,9 +121,15 @@ export function httpErrorCode(status: number, error?: WireError['error']): strin
*/
export class DeepSeekAdapter extends LlmAdapter {
private readonly streamIdleTimeoutMs: number
private readonly retryPolicy: ResolvedRetryPolicy
constructor(private readonly options: DeepSeekAdapterOptions) {
super()
if (options.defaults?.thinking === 'disabled'
&& options.defaults.reasoningEffort !== undefined
&& options.defaults.reasoningEffort !== 'off') {
throw new Error('llm-deepseek: only reasoningEffort "off" can be configured when thinking is disabled')
}
if (options.defaultContextWindow !== undefined
&& (!Number.isInteger(options.defaultContextWindow) || options.defaultContextWindow <= 0)) {
throw new Error('llm-deepseek: defaultContextWindow must be a positive integer')
@@ -110,30 +142,52 @@ export class DeepSeekAdapter extends LlmAdapter {
`llm-deepseek: streamIdleTimeoutMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`,
)
}
this.retryPolicy = resolveRetryPolicy(options.retryPolicy, 'llm-deepseek: retryPolicy')
}
override providerInfo(provider: string): LlmProviderInfo {
return { id: provider, name: 'DeepSeek' }
}
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
return Promise.resolve((this.options.models ?? []).map(model => ({
provider,
id: model.id,
name: model.name ?? model.id,
...model.description === undefined ? {} : { description: model.description },
inputModalities: ['text'],
outputModalities: ['text'],
})))
override providerRetryPolicy(_provider: string): ResolvedRetryPolicy {
return this.retryPolicy
}
override resolveModelContext(
_provider: string,
override listModels(provider: string): Promise<readonly LlmModelInfo[]> {
return Promise.resolve((this.options.models ?? []).map(model => modelInfo(provider, model)))
}
override resolveModel(
provider: string,
model: string,
): Promise<LlmModelContext | undefined> {
const contextWindow = this.options.models?.find(entry => entry.id === model)?.contextWindow
_signal?: AbortSignal,
): Promise<LlmResolvedModelInfo> {
const configured = this.options.models?.find(entry => entry.id === model)
const contextWindow = configured?.contextWindow
?? this.options.defaultContextWindow
return Promise.resolve(contextWindow === undefined ? undefined : { contextWindow })
return Promise.resolve({
...configured === undefined
? { provider, id: model, name: model }
: modelInfo(provider, configured),
...contextWindow === undefined ? {} : { context: { contextWindow } },
...this.options.defaults?.thinking === 'disabled'
? {
reasoning: {
efforts: OFF_ONLY_REASONING_EFFORTS,
defaultEffort: OFF_REASONING_EFFORT,
},
}
: {
reasoning: {
efforts: REASONING_EFFORTS,
defaultEffort: this.options.defaults?.reasoningEffort === 'off'
? OFF_REASONING_EFFORT
: this.options.defaults?.reasoningEffort === 'max'
? MAX_REASONING_EFFORT
: HIGH_REASONING_EFFORT,
},
},
})
}
async * stream(options: GenerateOptions): AsyncIterable<StreamChunk> {

View File

@@ -7,7 +7,8 @@
import type { Context } from 'cordis'
import z from 'schemastery'
import type {} from '@deepseek-ai/dsh-llm'
import { RetryPolicySchema } from '@deepseek-ai/dsh-llm'
import type { RetryPolicyConfig } from '@deepseek-ai/dsh-llm'
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
import { DEFAULT_STREAM_IDLE_TIMEOUT_MS, DeepSeekAdapter } from './adapter.ts'
import type { DeepSeekCatalogModel } from './adapter.ts'
@@ -21,31 +22,34 @@ export const name = 'llm-deepseek'
export const inject = ['llm']
const DEFAULT_MODELS: DeepSeekCatalogModel[] = [
{ id: 'deepseek-v4-flash', contextWindow: 128_000 },
{ id: 'deepseek-v4-pro', contextWindow: 128_000 },
{ id: 'deepseek-v4-flash', name: 'DeepSeek-V4-Flash', contextWindow: 256_000 },
{ id: 'deepseek-v4-pro', name: 'DeepSeek-V4-Pro', contextWindow: 256_000 },
]
/**
* Plugin config, validated by the same-named schemastery schema. Every field
* is optional in yml: credentials/endpoint fall back to the environment (a
* missing API key fails plugin load, not the first call), and omitted
* thinking fields send nothing on the wire, so the provider default applies.
* missing API key fails plugin load, not the first call), omitted thinking
* mode uses the provider default, and omitted reasoning effort resolves to
* `high`.
*/
export interface Config {
/** API key; falls back to $DEEPSEEK_API_KEY. Required one way or the other. */
apiKey?: string
/** Endpoint base; falls back to $DEEPSEEK_BASE_URL, then the public API. */
baseURL?: string
/** Thinking-mode default for every request (provider default: enabled). */
/** Deployment thinking policy; `disabled` limits every conversation request to `off`. */
thinking?: 'enabled' | 'disabled'
/** Thinking effort (only meaningful with thinking enabled). */
reasoningEffort?: 'high' | 'max'
/** Default thinking effort (default `high`); `off` disables thinking per request. */
reasoningEffort?: 'off' | 'high' | 'max'
/** Positive context capacity used when the selected model has no exact value. */
defaultContextWindow?: number
/** Advisory models shown by discovery consumers; defaults to V4 Flash and V4 Pro. */
models?: DeepSeekCatalogModel[]
/** Maximum provider idle time while one stream read is outstanding (default five minutes). */
streamIdleTimeoutMs?: number
/** Provider-owned model-request retry policy; omission uses normal defaults. */
retryPolicy?: RetryPolicyConfig
}
const catalogModel: z<DeepSeekCatalogModel> = z.object({
@@ -59,10 +63,11 @@ export const Config: z<Config> = z.object({
apiKey: z.string(),
baseURL: z.string(),
thinking: z.union(['enabled', 'disabled']),
reasoningEffort: z.union(['high', 'max']),
reasoningEffort: z.union(['off', 'high', 'max']),
defaultContextWindow: z.number().step(1).min(1),
models: z.array(catalogModel).default(DEFAULT_MODELS),
streamIdleTimeoutMs: z.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
retryPolicy: RetryPolicySchema,
})
/** Public API default; the internal endpoint comes from $DEEPSEEK_BASE_URL. */
@@ -94,6 +99,11 @@ function resolveModels(models: readonly DeepSeekCatalogModel[] | undefined): Dee
}
export function apply(ctx: Context, config: Config): void {
if (config.thinking === 'disabled'
&& config.reasoningEffort !== undefined
&& config.reasoningEffort !== 'off') {
throw new Error('llm-deepseek: only reasoningEffort "off" can be configured when thinking is disabled')
}
const apiKey = config.apiKey ?? process.env.DEEPSEEK_API_KEY
if (apiKey === undefined || apiKey.length === 0) {
throw new Error('llm-deepseek: an API key is required (Config.apiKey or $DEEPSEEK_API_KEY)')
@@ -111,5 +121,6 @@ export function apply(ctx: Context, config: Config): void {
: { defaultContextWindow: config.defaultContextWindow },
models: resolveModels(config.models),
streamIdleTimeoutMs: config.streamIdleTimeoutMs ?? DEFAULT_STREAM_IDLE_TIMEOUT_MS,
...config.retryPolicy === undefined ? {} : { retryPolicy: config.retryPolicy },
}))
}

View File

@@ -14,7 +14,42 @@ import type { WireMessage, WireRequest, WireTool } from './types.ts'
/** Adapter-level request defaults (from plugin config). */
export interface RequestDefaults {
thinking?: 'enabled' | 'disabled' | undefined
reasoningEffort?: 'high' | 'max' | undefined
reasoningEffort?: 'off' | 'high' | 'max' | undefined
}
interface ResolvedThinking {
thinking?: 'enabled' | 'disabled'
reasoningEffort?: 'high' | 'max'
}
/** Validate the adapter-owned effort before resolving its DeepSeek wire fields. */
function reasoningEffort(effort: NonNullable<GenerateOptions['reasoningEffort']>): 'off' | 'high' | 'max' {
if (effort === 'off' || effort === 'high' || effort === 'max') {
return effort as 'off' | 'high' | 'max'
}
throw new LlmError(
`DeepSeek does not support reasoning effort "${effort}"`,
'UNSUPPORTED_REASONING_EFFORT',
)
}
/** Resolve one legal thinking/effort pair without exposing `off` as a wire effort. */
function resolveThinking(options: GenerateOptions, defaults: RequestDefaults): ResolvedThinking {
if (options.purpose === 'session-title') return { thinking: 'disabled' }
const effort = options.reasoningEffort === undefined
? defaults.reasoningEffort
: reasoningEffort(options.reasoningEffort)
if (defaults.thinking === 'disabled' && effort !== undefined && effort !== 'off') {
throw new LlmError(
`DeepSeek deployment does not support reasoning effort "${effort}"`,
'UNSUPPORTED_REASONING_EFFORT',
)
}
if (effort === 'off') return { thinking: 'disabled' }
if (effort === 'high' || effort === 'max') {
return { thinking: 'enabled', reasoningEffort: effort }
}
return defaults.thinking === undefined ? {} : { thinking: defaults.thinking }
}
/** Join the text blocks of a message (used for user/tool-result content). */
@@ -133,16 +168,17 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa
}))
// A short title budget must produce visible text; conversation and
// compaction calls continue to inherit the adapter's thinking defaults.
const thinking = options.purpose === 'session-title' ? 'disabled' : defaults.thinking
const reasoningEffort = options.purpose === 'session-title' ? undefined : defaults.reasoningEffort
const resolvedThinking = resolveThinking(options, defaults)
return {
model: options.model,
messages,
stream: true,
stream_options: { include_usage: true },
...thinking !== undefined ? { thinking: { type: thinking } } : {},
...reasoningEffort !== undefined ? { reasoning_effort: reasoningEffort } : {},
...resolvedThinking.thinking !== undefined ? { thinking: { type: resolvedThinking.thinking } } : {},
...resolvedThinking.reasoningEffort !== undefined
? { reasoning_effort: resolvedThinking.reasoningEffort }
: {},
...tools !== undefined && tools.length > 0 ? { tools } : {},
...options.temperature !== undefined ? { temperature: options.temperature } : {},
...options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {},

View File

@@ -1,65 +1,33 @@
/**
* Decode an SSE byte stream into event `data` payloads. Network reads may split UTF-8 or lines;
* CRLF, comments, non-data fields, and multi-data events are handled per SSE rules. The literal
* `[DONE]` is yielded so the caller owns final flushing, and EOF before it raises {@link LlmError}.
* Decode an SSE byte stream into event `data` payloads. Framing — chunk
* reassembly, UTF-8/CRLF/BOM handling, comment and non-data field skipping,
* multi-`data:` joining — is `eventsource-parser`'s; this module keeps only
* the DeepSeek protocol: the literal `[DONE]` is yielded so the caller owns
* final flushing, and EOF before it raises {@link LlmError}. Framing is
* spec-strict: an event dispatches only on its blank-line terminator, so an
* unterminated tail at EOF is truncation, not a flushable payload.
*
* Minimal SSE (text/event-stream) parser for the chat-completions stream.
* @module dsh-llm-deepseek/sse
*/
import { EventSourceParserStream } from 'eventsource-parser/stream'
import { LlmError } from '@deepseek-ai/dsh-llm'
/** The terminal payload DeepSeek (and OpenAI) send after the last chunk. */
export const DONE = '[DONE]'
/** Extract the joined data payload from one raw SSE event block. */
function eventData(block: string): string | undefined {
const data: string[] = []
for (const rawLine of block.split('\n')) {
const line = rawLine.endsWith('\r') ? rawLine.slice(0, -1) : rawLine
if (line.startsWith('data:')) {
// The spec strips ONE leading space after the colon.
data.push(line.startsWith('data: ') ? line.slice(6) : line.slice(5))
}
// Comments (':…') and other fields (event:, id:, retry:) are ignored.
}
if (data.length === 0) return undefined
return data.join('\n')
}
/**
* Parse a byte stream into SSE data payloads. Yields `[DONE]` as the final
* Parse an SSE byte stream into data payloads. Yields `[DONE]` as the final
* value and returns; throws `LlmError('STREAM_CLOSED')` when the stream ends
* without it (truncated response — the model call cannot be trusted).
* @param stream - raw SSE bytes; reads may split anywhere, including mid-UTF-8 sequence.
* @returns each event's data payload in arrival order, the `[DONE]` sentinel last.
*/
export async function* parseSse(stream: AsyncIterable<Uint8Array>): AsyncGenerator<string> {
const decoder = new TextDecoder()
let buffer = ''
for await (const bytes of stream) {
buffer += decoder.decode(bytes, { stream: true })
// Events are separated by a blank line (\n\n; tolerate \r\n\r\n via the
// per-line \r strip in eventData and a normalized split here).
let boundary: number
while ((boundary = buffer.search(/\r?\n\r?\n/)) !== -1) {
const matched = /\r?\n\r?\n/.exec(buffer.slice(boundary))
const block = buffer.slice(0, boundary)
// matched cannot be null: search() just found the same pattern at 0.
buffer = buffer.slice(boundary + (matched as RegExpExecArray)[0].length)
const data = eventData(block)
if (data === undefined) continue
yield data
if (data === DONE) return
}
}
// Flush any final un-terminated event (servers usually end with \n\n, but
// a trailing block without one is still parseable).
buffer += decoder.decode()
const data = eventData(buffer)
if (data !== undefined) {
export async function* parseSse(stream: ReadableStream<BufferSource>): AsyncGenerator<string> {
const events = stream
.pipeThrough(new TextDecoderStream())
.pipeThrough(new EventSourceParserStream())
for await (const { data } of events) {
yield data
if (data === DONE) return
}