Merge remote-tracking branch 'origin/master' into codex/provider-retry-policy

# Conflicts:
#	.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md
#	docs/core-data-structures/llm-streaming.i18n.yaml
#	packages/llm/llm-retry/README.md
#	packages/llm/llm-retry/src/index.ts
#	packages/llm/llm-retry/tests/retry.spec.ts
This commit is contained in:
Turtle
2026-07-25 16:05:05 +08:00
156 changed files with 1412 additions and 1079 deletions

View File

@@ -55,7 +55,7 @@ Every request carries the shared attribution header from dsh-llm's `attributionH
## Errors
Non-2xx responses throw `LlmError` with stable codes: `AUTH` (401/403), `QUOTA` (a response whose provider details identify exhausted quota, balance, or credits), `RATE_LIMIT` (other 429s), `CONTEXT_WINDOW_EXCEEDED` (a 400 whose provider code, type, or message identifies context overflow), `INVALID_REQUEST` (other 400s), `SERVER` (5xx), `HTTP_<status>` otherwise. Its serializable `failure` retains the HTTP status plus a valid positive `Retry-After` seconds/date delay and `x-request-id` / `x-deepseek-request-id` when present. A pre-response transport failure (DNS, refused connection, TLS, proxy) throws `TRANSPORT` naming the configured endpoint and chaining the original rejection as `cause`; caller aborts throw `ABORTED`, and the loop's cancellation signal remains authoritative. Protocol violations throw `STREAM_CLOSED` (no `[DONE]`) or `MALFORMED_RESPONSE` (bad JSON payload). Unknown wire `finish_reason`s (e.g. `content_filter`, `insufficient_system_resource`) become `finish {kind: 'error', failure}` chunks.
Non-2xx responses throw `LlmError` with stable codes: `AUTH` (401/403), `QUOTA` (a response whose provider details identify exhausted quota, balance, or credits), `RATE_LIMIT` (other 429s), `CONTEXT_WINDOW_EXCEEDED` (a 400 whose provider code, type, or message identifies context overflow), `INVALID_REQUEST` (other 400s), `SERVER` (5xx), `HTTP_<status>` otherwise. Its serializable `failure` retains the HTTP status plus a valid positive `Retry-After` seconds/date delay and `x-request-id` / `x-deepseek-request-id` when present. A pre-response transport failure (DNS, refused connection, TLS, proxy) throws `TRANSPORT` naming the configured endpoint and chaining the original rejection as `cause`; caller aborts throw `ABORTED`, and the loop's cancellation signal remains authoritative. Protocol violations throw `STREAM_CLOSED` (no `[DONE]`) or `MALFORMED_RESPONSE` (bad JSON payload). Unknown wire `finish_reason`s (e.g. `content_filter`, `insufficient_system_resource`) become `finish {kind: 'error', failure}` chunks, and a completed stream whose `stop` (or absent) finish opened no content blocks becomes a `finish {kind: 'error'}` with code `EMPTY_RESPONSE` (retried by default policy).
## Testing

View File

@@ -8,7 +8,7 @@
* @module dsh-llm-deepseek/translate
*/
import { CallId, LlmError } from '@deepseek-ai/dsh-llm'
import { CallId, EMPTY_RESPONSE_CODE, LlmError } from '@deepseek-ai/dsh-llm'
import type { ContentBlock, FinishReason, StreamChunk, TokenUsage } from '@deepseek-ai/dsh-llm'
import { DONE } from './sse.ts'
import type { WireChunk, WireUsage } from './types.ts'
@@ -80,6 +80,8 @@ function closeBlock(block: OpenBlock): ContentBlock {
* Malformed JSON payloads abort the stream with `MALFORMED_RESPONSE`.
* @param payloads - SSE data payloads from {@link parseSse}, `[DONE]`-terminated.
* @returns deltas as they arrive; `block-end`s, `usage`, and `finish` are all deferred to the `[DONE]` sentinel.
* A `stop` (or absent) finish with no opened blocks is a degenerate provider completion and maps to an
* `EMPTY_RESPONSE` error finish instead of a successful empty message.
*/
export async function* translate(payloads: AsyncIterable<string>): AsyncGenerator<StreamChunk> {
let nextIndex = 0
@@ -102,7 +104,16 @@ export async function* translate(payloads: AsyncIterable<string>): AsyncGenerato
yield { type: 'block-end', index: block.index, block: closeBlock(block) }
}
if (pendingUsage) yield { type: 'usage', usage: pendingUsage }
yield { type: 'finish', reason: pendingFinish ?? { kind: 'stop' } }
const reason = pendingFinish ?? { kind: 'stop' as const }
yield {
type: 'finish',
reason: reason.kind === 'stop' && order.length === 0
? {
kind: 'error',
failure: { message: 'model returned a completed response with no content', code: EMPTY_RESPONSE_CODE },
}
: reason,
}
return
}

View File

@@ -1,5 +1,5 @@
import { describe, expect, it } from 'vitest'
import { BlockAssembler, LlmError } from '@deepseek-ai/dsh-llm'
import { BlockAssembler, EMPTY_RESPONSE_CODE, LlmError } from '@deepseek-ai/dsh-llm'
import type { StreamChunk } from '@deepseek-ai/dsh-llm'
import { DONE } from '../src/sse.ts'
import { mapFinishReason, mapUsage, translate } from '../src/translate.ts'
@@ -203,7 +203,50 @@ describe('translate: finish and usage handling', () => {
it('handles chunks with no choices at all', async () => {
const chunks = await collect(translate(feed({}, DONE)))
expect(chunks).toEqual([{ type: 'finish', reason: { kind: 'stop' } }])
expect(chunks).toEqual([{
type: 'finish',
reason: {
kind: 'error',
failure: { message: 'model returned a completed response with no content', code: EMPTY_RESPONSE_CODE },
},
}])
})
it('classifies an explicit stop with no opened blocks as EMPTY_RESPONSE, after usage', async () => {
const chunks = await collect(translate(feed(
firstChunk,
{ choices: [{ delta: {}, finish_reason: 'stop' }], usage: { prompt_tokens: 7, completion_tokens: 0 } },
DONE,
)))
expect(chunks).toEqual([
{ type: 'usage', usage: { inputTokens: 7, outputTokens: 0 } },
{
type: 'finish',
reason: {
kind: 'error',
failure: { message: 'model returned a completed response with no content', code: EMPTY_RESPONSE_CODE },
},
},
])
})
it('keeps a reasoning-only stream a successful stop (any opened block counts)', async () => {
const chunks = await collect(translate(feed(
firstChunk,
{ choices: [{ delta: { content: null, reasoning_content: 'mull' } }] },
{ choices: [{ delta: {}, finish_reason: 'stop' }] },
DONE,
)))
expect(chunks.at(-1)).toEqual({ type: 'finish', reason: { kind: 'stop' } })
})
it('leaves non-stop finishes unclassified even with no opened blocks', async () => {
const chunks = await collect(translate(feed(
firstChunk,
{ choices: [{ delta: {}, finish_reason: 'length' }] },
DONE,
)))
expect(chunks.at(-1)).toEqual({ type: 'finish', reason: { kind: 'max-tokens' } })
})
})

View File

@@ -52,7 +52,7 @@ If a listener rewrites assembled assistant content, the loop drops replay state
## Vocabulary differences
- pi-ai tool-call arguments are parsed objects; the harness stores raw JSON strings. The adapter parses input and re-stringifies output.
- pi-ai reports failures as in-stream error events; these map to `finish {kind:'error'|'aborted', failure}` chunks. Provider-specific error text distinguishes terminal `QUOTA` from transient `RATE_LIMIT`, while text and usage signals evaluated against the resolved model's context window normalize overflow to `CONTEXT_WINDOW_EXCEEDED`.
- pi-ai reports failures as in-stream error events; these map to `finish {kind:'error'|'aborted', failure}` chunks. Provider-specific error text distinguishes terminal `QUOTA` from transient `RATE_LIMIT`, while text and usage signals evaluated against the resolved model's context window normalize overflow to `CONTEXT_WINDOW_EXCEEDED`. A terminal `stop` whose message carries no content blocks maps to a `finish {kind:'error'}` with code `EMPTY_RESPONSE` (retried by default policy) instead of a successful empty message.
- pi-ai folds reasoning tokens into output usage; there is no separate reasoning count to map.
- `GenerateOptions.stop` is rejected with `UNSUPPORTED_OPTION` because pi-ai's common streaming surface cannot guarantee it across providers.
@@ -104,4 +104,4 @@ Recorded response content appends to the next request and does not invalidate it
- **`GenerateOptions.stop` is unsupported** — pi-ai's common stream options cannot guarantee stop-sequence behavior across providers, so the adapter rejects the field.
- **In-history `system` messages use pi-ai's common context conversion** — provider-specific placement follows pi-ai rather than a harness-owned wire override.
- **Provider HTTP status is unavailable** — pi-ai error events do not expose a stable HTTP status across providers; failures expose only stable harness error codes.
- **Retry policy is not an adapter option** — SDK retries are disabled so durable agent steps and `llm/retry` events own every visible attempt; direct `ctx.llm.stream()` calls remain single-attempt.
- **Retry policy is provider-owned, not an SDK retry** — each provider profile may configure nested `retryPolicy`, which `dsh-llm-retry` executes at the agent failed-step seam; pi-ai SDK retries stay disabled so durable agent steps and `llm/retry` events own every visible attempt, and direct `ctx.llm.stream()` calls remain single-attempt.

View File

@@ -8,7 +8,7 @@
* @module dsh-llm-pi-ai/stream
*/
import { CallId, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmError, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm'
import { CallId, CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, isContextWindowExceededError, isQuotaExceededError, LlmError, QUOTA_EXCEEDED_CODE } from '@deepseek-ai/dsh-llm'
import type { FinishReason, StreamChunk, TokenUsage } from '@deepseek-ai/dsh-llm'
import { isContextOverflow } from '@earendil-works/pi-ai'
import type { AssistantMessage, AssistantMessageEvent, Usage as PiUsage } from '@earendil-works/pi-ai'
@@ -48,7 +48,8 @@ function classifyPiAiError(message: string): string {
* @param contextWindow - resolved catalog capacity for usage-based overflow detection.
* @returns the mapped harness reason. Recognized error text, `stop` usage above
* `contextWindow`, and zero-output `length` usage that fills the window map
* to `CONTEXT_WINDOW_EXCEEDED`.
* to `CONTEXT_WINDOW_EXCEEDED`; a `stop` with no content blocks maps to an
* `EMPTY_RESPONSE` error.
*/
export function mapStopReason(message: AssistantMessage, contextWindow?: number): FinishReason {
const piAiOverflow = isContextOverflow(message, contextWindow)
@@ -66,7 +67,19 @@ export function mapStopReason(message: AssistantMessage, contextWindow?: number)
}
switch (message.stopReason) {
case 'stop': return { kind: 'stop' }
case 'stop':
// A terminal stop that produced no content blocks is a degenerate
// provider completion, not a successful (empty) assistant message.
if (message.content.length === 0) {
return {
kind: 'error',
failure: {
message: `model "${message.model}" returned a completed response with no content`,
code: EMPTY_RESPONSE_CODE,
},
}
}
return { kind: 'stop' }
case 'length': return { kind: 'max-tokens' }
case 'toolUse': return { kind: 'tool-calls' }
case 'aborted': return {

View File

@@ -1,5 +1,5 @@
import { describe, expect, it } from 'vitest'
import { CallId, CONTEXT_WINDOW_EXCEEDED_CODE, LlmError } from '@deepseek-ai/dsh-llm'
import { CallId, CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, LlmError } from '@deepseek-ai/dsh-llm'
import type { ContentBlock, StreamChunk } from '@deepseek-ai/dsh-llm'
import type { AssistantMessage, AssistantMessageEvent, Usage } from '@earendil-works/pi-ai'
import { toPiContext } from '../src/context.ts'
@@ -520,7 +520,22 @@ describe('mapStopReason / mapUsage', () => {
['toolUse', { kind: 'tool-calls' }],
['aborted', { kind: 'aborted', failure: { message: 'pi-ai stream aborted', code: 'ABORTED' } }],
] as const)('maps %s', (stopReason, expected) => {
expect(mapStopReason(assistant({ stopReason }))).toEqual(expected)
expect(mapStopReason(assistant({ stopReason, content: [{ type: 'text', text: 'ok' }] }))).toEqual(expected)
})
it('classifies a completed stop with no content as an EMPTY_RESPONSE error', () => {
expect(mapStopReason(assistant({ stopReason: 'stop' }))).toEqual({
kind: 'error',
failure: {
message: 'model "deepseek-v4-flash" returned a completed response with no content',
code: EMPTY_RESPONSE_CODE,
},
})
})
it('keeps a thinking-only stop successful (any block counts as content)', () => {
expect(mapStopReason(assistant({ stopReason: 'stop', content: [{ type: 'thinking', thinking: 'mull' }] })))
.toEqual({ kind: 'stop' })
})
it('defaults the error message when pi-ai omits it', () => {
@@ -580,7 +595,9 @@ describe('mapStopReason / mapUsage', () => {
})
it('uses the resolved context window for silent and length-stop overflows', () => {
const silent = assistant({ stopReason: 'stop', usage: usage(101, 0) })
// Non-empty content keeps the no-window branch on the successful stop path
// (an empty stop is EMPTY_RESPONSE, covered above); overflow wins over both.
const silent = assistant({ stopReason: 'stop', usage: usage(101, 0), content: [{ type: 'text', text: 'x' }] })
expect(mapStopReason(silent)).toEqual({ kind: 'stop' })
expect(mapStopReason(silent, 100)).toEqual({
kind: 'error',

View File

@@ -2,7 +2,7 @@
Function plugin that applies exact-provider retry policy on the agent loop's closed-step recovery seam. It does not wrap `ctx.llm.stream()`: every adapter call remains one provider attempt, and every retry opens a fresh numbered step.
Each provider adapter owns an optional nested `retryPolicy`, captured when its route registers on `ctx.llm` and carried with each call that reaches that registration's final adapter boundary. An in-flight failure retains that serving policy if the route is later disposed or replaced; a failure before any final adapter is selected has no provider policy and delegates. Omission uses normal mode: two retries for `RATE_LIMIT`, `SERVER`, `TIMEOUT`, and `TRANSPORT`. A normal policy can change its finite budget, eligible codes, and backoff. Always mode asks downstream recovery first, then retries every model-request failure without an attempt limit; success, cancellation, or plugin disposal stops it after active delegated recovery reaches quiescence.
Each provider adapter owns an optional nested `retryPolicy`, captured when its route registers on `ctx.llm` and carried with each call that reaches that registration's final adapter boundary. An in-flight failure retains that serving policy if the route is later disposed or replaced; a failure before any final adapter is selected has no provider policy and delegates. Omission uses normal mode: two retries for `EMPTY_RESPONSE`, `RATE_LIMIT`, `SERVER`, `TIMEOUT`, and `TRANSPORT`, with bounded exponential backoff from 500 ms to 10 seconds and 10 percent jitter. `EMPTY_RESPONSE` is the adapters' classification of a degenerate provider completion that produced no durable content, so repeating it is safe. A normal policy can change its finite budget, eligible codes, and backoff. Always mode asks downstream recovery first, then retries every model-request failure without an attempt limit; success, cancellation, or plugin disposal stops it after active delegated recovery reaches quiescence.
Both modes use bounded exponential backoff with symmetric jitter. A valid `providerRetryAfterMs` at or below `maxDelayMs` replaces local backoff without jitter. An over-cap provider delay makes normal mode delegate, while always mode uses its configured local backoff so it cannot terminate on that instruction.

View File

@@ -2,6 +2,7 @@
import type { Context } from 'cordis'
import type { Session, SessionEvent } from '@deepseek-ai/dsh-session'
import type { LlmFailure } from '@deepseek-ai/dsh-llm'
import type { InvariantFailure, InvariantInstaller } from '@deepseek-ai/dsh-invariants'
import { providerForClosedStep } from './history.ts'
import { parseRetryPolicyKey } from './policy-key.ts'
@@ -14,6 +15,32 @@ export const name = 'llm-retry-invariant'
/** Service required before the companion can reserve package ownership. */
export const inject = ['invariants']
/** Validate the complete provider-neutral failure payload at the durable boundary. */
function validateFailure(value: unknown, fail: InvariantFailure): asserts value is LlmFailure {
if (typeof value !== 'object' || value === null) {
fail('llm/retry failure must be an object')
}
const failure = value as Partial<LlmFailure>
if (typeof failure.message !== 'string' || failure.message.length === 0) {
fail('llm/retry failure.message must be a non-empty string')
}
if (typeof failure.code !== 'string' || failure.code.length === 0) {
fail('llm/retry failure.code must be a non-empty string')
}
if (failure.status !== undefined
&& (!Number.isInteger(failure.status) || failure.status < 100 || failure.status > 599)) {
fail('llm/retry failure.status must be an integer from 100 through 599 when present')
}
if (failure.providerRetryAfterMs !== undefined
&& (!Number.isFinite(failure.providerRetryAfterMs) || failure.providerRetryAfterMs <= 0)) {
fail('llm/retry failure.providerRetryAfterMs must be a positive finite number when present')
}
if (failure.requestId !== undefined
&& (typeof failure.requestId !== 'string' || failure.requestId.length === 0)) {
fail('llm/retry failure.requestId must be a non-empty string when present')
}
}
/** Validate one retry record against the open turn and most recently closed step. */
function validateRetry(
history: readonly SessionEvent[],
@@ -21,6 +48,8 @@ function validateRetry(
fail: InvariantFailure,
): void {
const { turn, step, provider, mode, policyKey, retry, delayMs } = event.data
const failure: unknown = event.data.failure
validateFailure(failure, fail)
if (!Number.isSafeInteger(retry) || retry < 1) {
fail('llm/retry retry must be a positive safe integer')
}
@@ -43,8 +72,8 @@ function validateRetry(
if (keyedPolicy.maxRetries !== maxRetries) {
fail(`llm/retry maxRetries ${maxRetries} must match policyKey`)
}
if (!keyedPolicy.retryableCodes.includes(event.data.failure.code)) {
fail(`llm/retry failure code ${event.data.failure.code} must be eligible under policyKey`)
if (!keyedPolicy.retryableCodes.includes(failure.code)) {
fail(`llm/retry failure code ${failure.code} must be eligible under policyKey`)
}
break
}

View File

@@ -1,6 +1,7 @@
import { describe, expect, it } from 'vitest'
import { Context } from 'cordis'
import SessionStore, { SessionId } from '@deepseek-ai/dsh-session'
import { ProviderRequestId } from '@deepseek-ai/dsh-llm'
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
import InvariantService from '@deepseek-ai/dsh-invariants'
import * as RetryInvariant from '@deepseek-ai/dsh-llm-retry/invariant'
@@ -105,6 +106,75 @@ describe('llm-retry invariants', () => {
}).toThrow(/always mode must omit maxRetries/)
})
it('validates complete durable failures before either retry mode uses them', async () => {
const ctx = await setup()
const complete = closeStep(ctx, 'retry-invariant-complete-failure')
expect(() => {
complete.append('llm/retry', {
turn: 1,
step: 1,
provider: 'mock',
mode: 'always',
policyKey: alwaysPolicyKey,
retry: 1,
delayMs: 1,
failure: {
message: 'provider busy',
code: 'RATE_LIMIT',
status: 429,
providerRetryAfterMs: 25,
requestId: ProviderRequestId('request-1'),
},
})
}).not.toThrow()
const normalNull = closeStep(ctx, 'retry-invariant-normal-null-failure')
expect(() => {
normalNull.append('llm/retry', {
turn: 1, step: 1, ...normal,
retry: 1, maxRetries: 2, delayMs: 1, failure: null,
} as never)
}).toThrow(/failure must be an object/)
const invalidFailures: readonly [string, unknown, RegExp][] = [
['always-null', null, /failure must be an object/],
['message-type', { message: 1, code: 'RATE_LIMIT' }, /failure\.message/],
['message-empty', { message: '', code: 'RATE_LIMIT' }, /failure\.message/],
['code-type', { message: 'failed', code: 1 }, /failure\.code/],
['code-empty', { message: 'failed', code: '' }, /failure\.code/],
['status-type', { message: 'failed', code: 'RATE_LIMIT', status: 429.5 }, /failure\.status/],
['status-low', { message: 'failed', code: 'RATE_LIMIT', status: 99 }, /failure\.status/],
['status-high', { message: 'failed', code: 'RATE_LIMIT', status: 600 }, /failure\.status/],
[
'retry-after-type',
{ message: 'failed', code: 'RATE_LIMIT', providerRetryAfterMs: '25' },
/failure\.providerRetryAfterMs/,
],
[
'retry-after-zero',
{ message: 'failed', code: 'RATE_LIMIT', providerRetryAfterMs: 0 },
/failure\.providerRetryAfterMs/,
],
['request-id-type', { message: 'failed', code: 'RATE_LIMIT', requestId: 1 }, /failure\.requestId/],
['request-id-empty', { message: 'failed', code: 'RATE_LIMIT', requestId: '' }, /failure\.requestId/],
]
for (const [name, invalidFailure, message] of invalidFailures) {
const session = closeStep(ctx, `retry-invariant-${name}`)
expect(() => {
session.append('llm/retry', {
turn: 1,
step: 1,
provider: 'mock',
mode: 'always',
policyKey: alwaysPolicyKey,
retry: 1,
delayMs: 1,
failure: invalidFailure,
} as never)
}).toThrow(message)
}
})
it('binds event mode and finite budget to the canonical policy key', async () => {
const ctx = await setup()
const normalModeMismatch = closeStep(ctx, 'retry-invariant-normal-mode-key')

View File

@@ -1,7 +1,7 @@
import { afterEach, describe, expect, expectTypeOf, it, vi } from 'vitest'
import { Context } from 'cordis'
import type { Fiber } from 'cordis'
import LlmService, { CallId, LlmAdapter, LlmError, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
import LlmService, { CallId, EMPTY_RESPONSE_CODE, LlmAdapter, LlmError, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
import type {
AlwaysRetryPolicyConfig,
BackoffConfig,
@@ -74,6 +74,25 @@ function textResponse(text: string): StreamChunk[] {
]
}
/**
* A degenerate empty provider completion as an error finish chunk. Both
* adapters emit this shape and the EMPTY_RESPONSE code (the field the policy
* routes on); the message text here is the deepseek adapter's phrasing (pi-ai
* qualifies it with the model name).
*/
function emptyCompletion(): StreamChunk[] {
return [
{ type: 'usage', usage: { inputTokens: 0, outputTokens: 0 } },
{
type: 'finish',
reason: {
kind: 'error',
failure: { message: 'model returned a completed response with no content', code: EMPTY_RESPONSE_CODE },
},
},
]
}
async function harness(
adapter: ScriptedAdapter,
policies: Readonly<Record<string, RetryPolicyConfig | undefined>> = { mock: normalConfig() },
@@ -184,7 +203,7 @@ describe('provider-routed retry policy', () => {
step: 1,
provider: 'mock',
mode: 'normal',
policyKey: '["normal",2,["RATE_LIMIT","SERVER","TIMEOUT","TRANSPORT"],500,10000,0.1]',
policyKey: '["normal",2,["EMPTY_RESPONSE","RATE_LIMIT","SERVER","TIMEOUT","TRANSPORT"],500,10000,0.1]',
retry: 1,
maxRetries: 2,
delayMs: 500,
@@ -247,6 +266,39 @@ describe('provider-routed retry policy', () => {
})
})
it('retries an EMPTY_RESPONSE error finish under the default retryable codes', async () => {
vi.useFakeTimers()
const adapter = new ScriptedAdapter([
emptyCompletion(),
textResponse('recovered'),
])
// No retryableCodes override: this proves the default policy covers the
// adapters' empty-completion classification end to end (finish-chunk error
// delivery, not a thrown stream error).
;({ ctx: context } = await harness(adapter))
const agent = context.agentLoop.create(SessionId('retry-empty-response'), { provider: 'mock', model: 'mock' })
const scheduled = waitForRetry(context, agent, 1)
agent.followup([{ type: 'text', text: 'go' }])
const event = await scheduled
expect(event.data.failure).toEqual({
message: 'model returned a completed response with no content',
code: EMPTY_RESPONSE_CODE,
})
const idle = waitForIdle(context, agent)
await vi.advanceTimersByTimeAsync(500)
await idle
expect(adapter.requests).toHaveLength(2)
expect(agent.session.events.filter(event => event.type === 'assistant/message').map(event => event.data.step))
.toEqual([2])
expect(agent.session.deriveMessages().at(-1)).toMatchObject({
role: 'assistant',
content: [{ type: 'text', text: 'recovered' }],
})
})
it('leaves partial failed chunks on their step without committing a message or tool side effect', async () => {
vi.useFakeTimers()
const adapter = new ScriptedAdapter([

View File

@@ -55,6 +55,7 @@ Every product adapter sends application identity on provider HTTP requests. `att
- `errorChain(value)` — renders a thrown value with its full `cause` chain and AggregateError members for diagnostic surfaces (UI notices, logger lines, durable `turn/end` messages), so transport wrappers like undici's `TypeError: fetch failed` surface the underlying `ECONNREFUSED`/DNS/TLS detail instead of masking it. Rendering only — route on `code`, never by parsing the result.
- `CONTEXT_WINDOW_EXCEEDED_CODE` — the provider-neutral code both DeepSeek adapters use when a request exceeds the model context window, regardless of thrown-HTTP versus in-band finish delivery. `isContextWindowExceededError(detail)` is their shared conservative classifier for OpenAI-compatible provider detail.
- `QUOTA_EXCEEDED_CODE` — the non-transient provider-neutral code for exhausted account quota, balance, credits, budget, or usage limits. `isQuotaExceededError(detail)` keeps those failures distinct from request-rate limits.
- `EMPTY_RESPONSE_CODE` — the provider-neutral code both adapters use for a degenerate provider completion: a terminal `stop` that carried no content blocks at all. Classified as an error finish (not a successful empty message) because the attempt produced nothing durable; `dsh-llm-retry` retries it by default.
### Real adapters

View File

@@ -27,6 +27,17 @@ export const CONTEXT_WINDOW_EXCEEDED_CODE = 'CONTEXT_WINDOW_EXCEEDED'
/** Canonical provider-neutral code for an exhausted account quota or balance. */
export const QUOTA_EXCEEDED_CODE = 'QUOTA'
/**
* Canonical provider-neutral code for a response that completed normally but
* carried no content blocks at all. Providers occasionally emit a degenerate
* completion (a terminal stop with zero output); adapters classify it as this
* failure instead of yielding an empty assistant message, because an empty
* message silently ends the turn with nothing for the user or the loop to act
* on. The attempt produced nothing durable, so retry policy treats it as safe
* to repeat.
*/
export const EMPTY_RESPONSE_CODE = 'EMPTY_RESPONSE'
/** Structured codes and plain phrases that explicitly name a context bound being exceeded. */
const STRUCTURED_CONTEXT_OVERFLOW = new RegExp(
String.raw`(?:^|[^a-z0-9])context[\s_-](?:length|window)[\s_-]`

View File

@@ -9,12 +9,19 @@
import z from 'schemastery'
import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout'
import { EMPTY_RESPONSE_CODE } from './error.ts'
const DEFAULT_MAX_RETRIES = 2
const DEFAULT_INITIAL_DELAY_MS = 500
const DEFAULT_MAX_DELAY_MS = 10_000
const DEFAULT_JITTER_RATIO = 0.1
const DEFAULT_RETRYABLE_CODES = Object.freeze(['RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'])
const DEFAULT_RETRYABLE_CODES = Object.freeze([
EMPTY_RESPONSE_CODE,
'RATE_LIMIT',
'SERVER',
'TIMEOUT',
'TRANSPORT',
])
/** Bounded exponential backoff with symmetric jitter around each local delay. */
export interface BackoffConfig {
@@ -159,7 +166,7 @@ export function resolveRetryPolicy(
if (retryableCodes.length === 0) {
throw new Error(`${path}.retryableCodes must not be empty`)
}
if (retryableCodes.some(code => code.length === 0)) {
if (retryableCodes.some(code => typeof code !== 'string' || code.length === 0)) {
throw new Error(`${path}.retryableCodes must contain only non-empty strings`)
}
if (new Set(retryableCodes).size !== retryableCodes.length) {

View File

@@ -13,7 +13,7 @@ describe('provider retry policy', () => {
expect(policy).toEqual({
mode: 'normal',
maxRetries: 2,
retryableCodes: ['RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'],
retryableCodes: ['EMPTY_RESPONSE', 'RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'],
initialDelayMs: 500,
maxDelayMs: 10_000,
jitterRatio: 0.1,
@@ -72,6 +72,7 @@ describe('provider retry policy', () => {
[{ mode: 'normal', retryableCodes: [] }, /must not be empty/],
[{ mode: 'normal', retryableCodes: ['SERVER', 'SERVER'] }, /duplicates/],
[{ mode: 'normal', retryableCodes: [''] }, /non-empty strings/],
[{ mode: 'normal', retryableCodes: [429] }, /non-empty strings/],
[{ mode: 'normal', maxRetires: 1 }, /unknown key "maxRetires"/],
[{ mode: 'always', maxRetries: 1 }, /unknown key "maxRetries"/],
[{ mode: 'always', backoff: { initialDelay: 1 } }, /unknown key "initialDelay"/],