feat: implement bounded LLM request recovery

This commit is contained in:
Tianyi Cui
2026-07-20 03:34:19 +08:00
parent 7cf966fc0e
commit 3b0b0cefeb
115 changed files with 3311 additions and 366 deletions

View File

@@ -2,7 +2,14 @@ import { createServer } from 'node:http'
import type { IncomingMessage, Server, ServerResponse } from 'node:http'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { Context } from 'cordis'
import LlmService, { CONTEXT_WINDOW_EXCEEDED_CODE, LlmError, userAgent } from '@deepseek-ai/dsh-llm'
import LlmService, {
CONTEXT_WINDOW_EXCEEDED_CODE,
LlmError,
ProviderRequestId,
QUOTA_EXCEEDED_CODE,
userAgent,
} from '@deepseek-ai/dsh-llm'
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
import { SessionId } from '@deepseek-ai/dsh-session'
import * as LlmDeepSeek from '@deepseek-ai/dsh-llm-deepseek'
import { DeepSeekAdapter } from '@deepseek-ai/dsh-llm-deepseek'
@@ -12,7 +19,7 @@ import { assemble } from './assemble.ts'
/** One scripted behavior for the next request the mock server receives. */
type Behavior =
| { kind: 'sse'; events: string[]; delayMs?: number }
| { kind: 'http-error'; status: number; body: string; contentType?: string }
| { kind: 'http-error'; status: number; body: string; contentType?: string; headers?: Record<string, string> }
| { kind: 'close-early'; events: string[] }
interface MockServer {
@@ -30,6 +37,7 @@ const servers: Server[] = []
afterEach(async () => {
await Promise.all(servers.splice(0).map(server => new Promise(resolve => server.close(resolve))))
vi.unstubAllEnvs()
vi.useRealTimers()
})
/** Local chat-completions stand-in: replays scripted behaviors per request. */
@@ -48,7 +56,10 @@ async function mockServer(script: Behavior[]): Promise<MockServer> {
return
}
if (behavior.kind === 'http-error') {
response.writeHead(behavior.status, { 'content-type': behavior.contentType ?? 'application/json' })
response.writeHead(behavior.status, {
'content-type': behavior.contentType ?? 'application/json',
...behavior.headers,
})
response.end(behavior.body)
return
}
@@ -202,6 +213,84 @@ describe('DeepSeekAdapter against a mock server', () => {
expect(code).toBe(CONTEXT_WINDOW_EXCEEDED_CODE)
})
it('retains status, Retry-After seconds, and provider request id as structured facts', async () => {
const server = await mockServer([{
kind: 'http-error',
status: 429,
body: JSON.stringify({ error: { message: 'slow down' } }),
headers: { 'retry-after': '2', 'x-request-id': 'req-429' },
}])
const ctx = await harness(server.url)
let thrown: unknown
try {
await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
} catch (error: unknown) {
thrown = error
}
expect(thrown).toBeInstanceOf(LlmError)
expect((thrown as LlmError).failure).toEqual({
message: 'slow down',
code: 'RATE_LIMIT',
status: 429,
retryAfterMs: 2_000,
requestId: ProviderRequestId('req-429'),
})
})
it('parses a future Retry-After HTTP date and the DeepSeek request-id fallback', async () => {
const now = 1_800_000_000_000
const dateNow = vi.spyOn(Date, 'now').mockReturnValue(now)
try {
const server = await mockServer([{
kind: 'http-error',
status: 503,
body: JSON.stringify({ error: { message: 'come back later' } }),
headers: {
'retry-after': new Date(now + 3_000).toUTCString(),
'x-deepseek-request-id': 'deepseek-503',
},
}])
const ctx = await harness(server.url)
await expect(assemble(ctx, { model: 'deepseek-v4-flash', messages: [] }))
.rejects.toMatchObject({
failure: {
message: 'come back later',
code: 'SERVER',
status: 503,
retryAfterMs: 3_000,
requestId: ProviderRequestId('deepseek-503'),
},
})
} finally {
dateNow.mockRestore()
}
})
it('omits zero, non-finite, invalid, and past Retry-After values', async () => {
const values = [
'0',
'9'.repeat(400),
'not-a-date',
new Date(0).toUTCString(),
]
for (const value of values) {
const server = await mockServer([{
kind: 'http-error',
status: 429,
body: JSON.stringify({ error: { message: 'retry later' } }),
headers: { 'retry-after': value },
}])
const ctx = await harness(server.url)
let thrown: LlmError | undefined
try {
await assemble(ctx, { model: 'deepseek-v4-flash', messages: [] })
} catch (error: unknown) {
if (error instanceof LlmError) thrown = error
}
expect(thrown?.failure).toEqual({ message: 'retry later', code: 'RATE_LIMIT', status: 429 })
}
})
it('classifies only context-capacity HTTP 400 details as context overflow', () => {
expect(httpErrorCode(400, { message: 'request too large for model context' }))
.toBe(CONTEXT_WINDOW_EXCEEDED_CODE)
@@ -210,6 +299,12 @@ describe('DeepSeekAdapter against a mock server', () => {
expect(httpErrorCode(413, { code: 'context_length_exceeded' })).toBe('HTTP_413')
})
it('distinguishes terminal quota exhaustion from transient HTTP 429 throttling', () => {
expect(httpErrorCode(429, { code: 'insufficient_quota', message: 'account credits exhausted' }))
.toBe(QUOTA_EXCEEDED_CODE)
expect(httpErrorCode(429, { message: 'request rate limit exceeded' })).toBe('RATE_LIMIT')
})
it('keeps the status-line message for JSON error bodies without a message', async () => {
const server = await mockServer([{ kind: 'http-error', status: 500, body: '{"error":{"type":"x"}}' }])
const ctx = await harness(server.url)
@@ -272,7 +367,76 @@ describe('DeepSeekAdapter against a mock server', () => {
})()
setTimeout(() => { controller.abort() }, 30)
await expect(pending).rejects.toThrow()
await expect(pending).rejects.toMatchObject({ code: 'ABORTED' })
})
it('maps connection failures to TRANSPORT without losing the cause', async () => {
const cause = new TypeError('connection refused')
const fetchSpy = vi.spyOn(globalThis, 'fetch').mockRejectedValue(cause)
const adapter = new DeepSeekAdapter({ apiKey: 'k', baseURL: 'https://example.invalid' })
try {
const drain = async (): Promise<void> => {
for await (const _chunk of adapter.stream({ provider: 'deepseek', model: 'm', messages: [] })) { /* drain */ }
}
await expect(drain()).rejects.toMatchObject({ code: 'TRANSPORT', cause })
} finally {
fetchSpy.mockRestore()
}
})
it('renders a non-Error transport rejection without losing its cause', async () => {
const fetchSpy = vi.spyOn(globalThis, 'fetch').mockImplementation(() => {
const failed = Promise.withResolvers<Response>()
failed.reject('offline')
return failed.promise
})
const adapter = new DeepSeekAdapter({ apiKey: 'k', baseURL: 'https://example.invalid' })
try {
const drain = async (): Promise<void> => {
for await (const _chunk of adapter.stream({ provider: 'deepseek', model: 'm', messages: [] })) { /* drain */ }
}
await expect(drain()).rejects.toMatchObject({
message: 'DeepSeek transport failed: offline',
code: 'TRANSPORT',
cause: 'offline',
})
} finally {
fetchSpy.mockRestore()
}
})
it('aborts the underlying body when the stream stays idle past its watchdog', async () => {
vi.useFakeTimers()
let stopped = false
const fetchSpy = vi.spyOn(globalThis, 'fetch').mockImplementation((_input, init) => {
const signal = init?.signal
const body = new ReadableStream<Uint8Array>({
start(controller) {
signal?.addEventListener('abort', () => {
stopped = true
controller.error(signal.reason)
}, { once: true })
},
})
return Promise.resolve(new Response(body, { status: 200 }))
})
const adapter = new DeepSeekAdapter({
apiKey: 'k',
baseURL: 'https://example.invalid',
streamIdleTimeoutMs: 100,
})
try {
const drain = (async () => {
for await (const _chunk of adapter.stream({ provider: 'deepseek', model: 'm', messages: [] })) { /* drain */ }
})()
const rejected = expect(drain).rejects.toMatchObject({ code: 'TIMEOUT' })
await vi.advanceTimersByTimeAsync(0)
await vi.advanceTimersByTimeAsync(100)
await rejected
expect(stopped).toBe(true)
} finally {
fetchSpy.mockRestore()
}
})
})
@@ -419,4 +583,30 @@ describe('plugin registration and config', () => {
expect(adapter).toBeInstanceOf(DeepSeekAdapter)
await expect(adapter.listModels('deepseek')).resolves.toEqual([])
})
it('rejects invalid idle watchdog bounds for direct and plugin composition', async () => {
expect(() => new DeepSeekAdapter({
apiKey: 'k',
baseURL: 'http://127.0.0.1:1',
streamIdleTimeoutMs: Number.POSITIVE_INFINITY,
})).toThrow(/streamIdleTimeoutMs.*positive finite/)
expect(() => new DeepSeekAdapter({
apiKey: 'k',
baseURL: 'http://127.0.0.1:1',
streamIdleTimeoutMs: MAX_TIMER_DELAY_MS + 1,
})).toThrow(/streamIdleTimeoutMs.*no greater/)
const ctx = new Context()
await ctx.plugin(LlmService)
await expect(ctx.plugin(LlmDeepSeek, {
apiKey: 'k',
baseURL: 'http://127.0.0.1:1',
streamIdleTimeoutMs: 0,
})).rejects.toThrow(/streamIdleTimeoutMs/)
await expect(ctx.plugin(LlmDeepSeek, {
apiKey: 'k',
baseURL: 'http://127.0.0.1:1',
streamIdleTimeoutMs: MAX_TIMER_DELAY_MS + 1,
})).rejects.toThrow(/streamIdleTimeoutMs/)
})
})

View File

@@ -232,8 +232,7 @@ describe('mapFinishReason', () => {
(wire) => {
expect(mapFinishReason(wire)).toEqual({
kind: 'error',
message: `model stopped: ${wire}`,
code: wire.toUpperCase(),
failure: { message: `model stopped: ${wire}`, code: wire.toUpperCase() },
})
},
)