feat(llm-deepseek): configure max token defaults

This commit is contained in:
Yichen Jiang
2026-07-30 21:04:00 +08:00
parent 2fb90a744e
commit daf70f3660
50 changed files with 430 additions and 95 deletions

View File

@@ -42,6 +42,8 @@ export interface DeepSeekAdapterOptions {
baseURL: string
/** Request defaults applied to every call (thinking mode, effort). */
defaults?: RequestDefaults
/** Default per-request output cap; explicit request values win. */
maxTokens?: number
/** Positive context capacity used when the selected model has no exact value. */
defaultContextWindow?: number
/** Advisory models exposed to discovery consumers; requests remain unrestricted. */
@@ -54,6 +56,10 @@ export interface DeepSeekAdapterOptions {
/** Default maximum idle interval while an adapter stream read is outstanding. */
export const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 300_000
/** Default combined request/response context capacity. */
export const DEFAULT_CONTEXT_WINDOW = 1_000_000
/** Default per-request output-token cap. */
export const DEFAULT_MAX_TOKENS = 256_000
const STREAM_IDLE_TIMEOUT_CODE = 'LLM_STREAM_IDLE_TIMEOUT'
const OFF_REASONING_EFFORT = ReasoningEffortId('off')
const HIGH_REASONING_EFFORT = ReasoningEffortId('high')
@@ -120,6 +126,8 @@ export function httpErrorCode(status: number, error?: WireError['error']): strin
export class DeepSeekAdapter extends LlmAdapter {
private readonly streamIdleTimeoutMs: number
private readonly retryPolicy: ResolvedRetryPolicy
private readonly defaultContextWindow: number
private readonly maxTokens: number
constructor(private readonly options: DeepSeekAdapterOptions) {
super()
@@ -128,10 +136,14 @@ export class DeepSeekAdapter extends LlmAdapter {
&& options.defaults.reasoningEffort !== 'off') {
throw new Error('llm-deepseek: only reasoningEffort "off" can be configured when thinking is disabled')
}
if (options.defaultContextWindow !== undefined
&& (!Number.isInteger(options.defaultContextWindow) || options.defaultContextWindow <= 0)) {
this.defaultContextWindow = options.defaultContextWindow ?? DEFAULT_CONTEXT_WINDOW
if (!Number.isInteger(this.defaultContextWindow) || this.defaultContextWindow <= 0) {
throw new Error('llm-deepseek: defaultContextWindow must be a positive integer')
}
this.maxTokens = options.maxTokens ?? DEFAULT_MAX_TOKENS
if (!Number.isSafeInteger(this.maxTokens) || this.maxTokens <= 0) {
throw new Error('llm-deepseek: maxTokens must be a positive safe integer')
}
this.streamIdleTimeoutMs = options.streamIdleTimeoutMs ?? DEFAULT_STREAM_IDLE_TIMEOUT_MS
if (!Number.isFinite(this.streamIdleTimeoutMs)
|| this.streamIdleTimeoutMs <= 0
@@ -162,12 +174,13 @@ export class DeepSeekAdapter extends LlmAdapter {
): Promise<LlmResolvedModelInfo> {
const configured = this.options.models?.find(entry => entry.id === model)
const contextWindow = configured?.contextWindow
?? this.options.defaultContextWindow
?? this.defaultContextWindow
return Promise.resolve({
...configured === undefined
? { provider, id: model, name: model }
: modelInfo(provider, configured),
...contextWindow === undefined ? {} : { context: { contextWindow } },
context: { contextWindow },
defaultMaxTokens: this.maxTokens,
...this.options.defaults?.thinking === 'disabled'
? {
reasoning: {
@@ -231,7 +244,7 @@ export class DeepSeekAdapter extends LlmAdapter {
}
private async * request(options: GenerateOptions, signal: AbortSignal): AsyncIterable<StreamChunk> {
const body = serializeRequest(options, this.options.defaults ?? {})
const body = serializeRequest(options, this.options.defaults, this.maxTokens)
// Prepared outside the try so the TRANSPORT label below covers exactly the
// transport boundary, never a serialization failure.
const payload = JSON.stringify(body)

View File

@@ -10,10 +10,20 @@ import z from 'schemastery'
import { RetryPolicySchema } from '@deepseek-ai/dsh-llm'
import type { RetryPolicyConfig } from '@deepseek-ai/dsh-llm'
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
import { DEFAULT_STREAM_IDLE_TIMEOUT_MS, DeepSeekAdapter } from './adapter.ts'
import {
DEFAULT_CONTEXT_WINDOW,
DEFAULT_MAX_TOKENS,
DEFAULT_STREAM_IDLE_TIMEOUT_MS,
DeepSeekAdapter,
} from './adapter.ts'
import type { DeepSeekCatalogModel } from './adapter.ts'
export { DeepSeekAdapter } from './adapter.ts'
export {
DEFAULT_CONTEXT_WINDOW,
DEFAULT_MAX_TOKENS,
DEFAULT_STREAM_IDLE_TIMEOUT_MS,
DeepSeekAdapter,
} from './adapter.ts'
export type { DeepSeekAdapterOptions, DeepSeekCatalogModel } from './adapter.ts'
export type { RequestDefaults } from './serialize.ts'
export type * from './types.ts'
@@ -22,8 +32,8 @@ export const name = 'llm-deepseek'
export const inject = ['llm']
const DEFAULT_MODELS: DeepSeekCatalogModel[] = [
{ id: 'deepseek-v4-flash', name: 'DeepSeek-V4-Flash', contextWindow: 256_000 },
{ id: 'deepseek-v4-pro', name: 'DeepSeek-V4-Pro', contextWindow: 256_000 },
{ id: 'deepseek-v4-flash', name: 'DeepSeek-V4-Flash', contextWindow: DEFAULT_CONTEXT_WINDOW },
{ id: 'deepseek-v4-pro', name: 'DeepSeek-V4-Pro', contextWindow: DEFAULT_CONTEXT_WINDOW },
]
/**
@@ -42,7 +52,9 @@ export interface Config {
thinking?: 'enabled' | 'disabled'
/** Default thinking effort (default `high`); `off` disables thinking per request. */
reasoningEffort?: 'off' | 'high' | 'max'
/** Positive context capacity used when the selected model has no exact value. */
/** Default per-request output cap (default 256,000); explicit request values win. */
maxTokens?: number
/** Positive context capacity used when the selected model has no exact value (default 1,000,000). */
defaultContextWindow?: number
/** Advisory models shown by discovery consumers; defaults to V4 Flash and V4 Pro. */
models?: DeepSeekCatalogModel[]
@@ -64,7 +76,8 @@ export const Config: z<Config> = z.object({
baseURL: z.string(),
thinking: z.union(['enabled', 'disabled']),
reasoningEffort: z.union(['off', 'high', 'max']),
defaultContextWindow: z.number().step(1).min(1),
maxTokens: z.number().step(1).min(1).max(Number.MAX_SAFE_INTEGER).default(DEFAULT_MAX_TOKENS),
defaultContextWindow: z.number().step(1).min(1).default(DEFAULT_CONTEXT_WINDOW),
models: z.array(catalogModel).default(DEFAULT_MODELS),
streamIdleTimeoutMs: z.number().min(Number.MIN_VALUE).max(MAX_TIMER_DELAY_MS).default(DEFAULT_STREAM_IDLE_TIMEOUT_MS),
retryPolicy: RetryPolicySchema,
@@ -116,9 +129,8 @@ export function apply(ctx: Context, config: Config): void {
thinking: config.thinking,
reasoningEffort: config.reasoningEffort,
},
...config.defaultContextWindow === undefined
? {}
: { defaultContextWindow: config.defaultContextWindow },
maxTokens: config.maxTokens ?? DEFAULT_MAX_TOKENS,
defaultContextWindow: config.defaultContextWindow ?? DEFAULT_CONTEXT_WINDOW,
models: resolveModels(config.models),
streamIdleTimeoutMs: config.streamIdleTimeoutMs ?? DEFAULT_STREAM_IDLE_TIMEOUT_MS,
...config.retryPolicy === undefined ? {} : { retryPolicy: config.retryPolicy },

View File

@@ -137,9 +137,14 @@ export function serializeMessages(messages: Message[]): WireMessage[] {
* provider defaults apply.
* @param options - the harness request (model, history, system, tools, sampling).
* @param defaults - adapter-level thinking defaults; undefined fields put nothing on the wire.
* @param defaultMaxTokens - adapter output default used only when the request omits a cap.
* @returns the chat-completions request body.
*/
export function serializeRequest(options: GenerateOptions, defaults: RequestDefaults = {}): WireRequest {
export function serializeRequest(
options: GenerateOptions,
defaults: RequestDefaults = {},
defaultMaxTokens?: number,
): WireRequest {
const messages: WireMessage[] = []
if (options.system !== undefined) {
messages.push({ role: 'system', content: options.system })
@@ -157,6 +162,7 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa
// A short title budget must produce visible text; conversation and
// compaction calls continue to inherit the adapter's thinking defaults.
const resolvedThinking = resolveThinking(options, defaults)
const maxTokens = options.maxTokens ?? defaultMaxTokens
return {
model: options.model,
@@ -169,7 +175,7 @@ export function serializeRequest(options: GenerateOptions, defaults: RequestDefa
: {},
...tools !== undefined && tools.length > 0 ? { tools } : {},
...options.temperature !== undefined ? { temperature: options.temperature } : {},
...options.maxTokens !== undefined ? { max_tokens: options.maxTokens } : {},
...maxTokens === undefined ? {} : { max_tokens: maxTokens },
...options.stop !== undefined ? { stop: options.stop } : {},
}
}