refactor(token-meter): make context occupancy durable projection state

Replace the transient `session/model-request` mux frame with ordinary durable
session state. Occupancy now rides two last-wins projection fields instead of a
non-replayable frame that needed removal tombstones and cross-stream fencing.

The frame was the only non-replayable class on the mux stream. Because host and
mux are independent SSE streams with no cross-stream order, a request emitted
before a removal could arrive after `host/session-removed`, and a legitimate
request for a new lifecycle reusing the same id could be fenced by a late
removal. Fixing that needed a lifecycle generation on every frame; the frame
itself was the problem.

Removed: the `session/model-request` frame and schema, the `agent/model-request`
core event, the ApiProxy measurement point, the client-side telemetry map and
removal tombstone, and the synthetic `cancelled` open error used to signal
reconnect through the error channel.

Added: `request/context`, a log-only session event recording the
registration-bound capacity of the route a request resolved to, appended beside
`request/header` from the lookup that already prepared the call and skipped when
the route is unchanged. Capacity stays out of `EpochHeader` because it is
adapter metadata about a route, not an input the request was built from, so it
must not join request reconstruction or header equality.

The `contextPressure` projection pairs the newest provider-reported prompt size
with the newest recorded capacity. The two are deliberately not one atomic
request observation: switching models can pair a fresh capacity with the prior
route's pressure until the next request reports usage. The figure is a
user-facing reference, and this matches how the TUI status line has always
computed occupancy.
This commit is contained in:
Hypatia May
2026-07-30 13:53:08 +08:00
parent fdccc58cef
commit 4819210142
62 changed files with 382 additions and 1362 deletions

View File

@@ -1,11 +1,11 @@
/**
* Pure fold for durable provider-reported token usage.
* Pure folds for durable provider-reported token usage and context occupancy.
*/
import { z } from 'zod'
import type { TokenUsage } from '@deepseek-ai/dsh-llm'
import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection'
import type { TokenUsageProjection } from './projection.ts'
import type { ContextPressureProjection, TokenUsageProjection } from './projection.ts'
interface UsageSample {
turn: number
@@ -56,13 +56,26 @@ const projectionSchema = z.object({
cacheWriteTokens: z.number().int().nonnegative(),
}).strict()
// Cast for the optional capacity: under exactOptionalPropertyTypes zod infers
// `number | undefined` where the interface declares an absent-or-number field.
const pressureSchema = z.object({
pressureTokens: z.number().int().nonnegative(),
contextWindow: z.number().int().positive().optional(),
}).strict() as unknown as z.ZodType<ContextPressureProjection>
/** Prompt-side pressure of one request: input plus cache traffic, no output. */
const pressureFrom = (usage: TokenUsage): number =>
usage.inputTokens + (usage.cacheReadTokens ?? 0) + (usage.cacheWriteTokens ?? 0)
/**
* Token-meter's session projection unit.
*
* Usage chunks provide an early sample that survives a later request failure;
* an assistant message provides the final sample for the same turn/step. A
* repeated sample replaces that step's earlier value instead of double
* counting it.
* counting it. The single `last` slot relies on the session-log invariant
* that usage reports for one turn/step are adjacent: once a later step begins,
* a legal log never reports usage for an earlier step again.
*/
export const tokenUsageProjectionDefinition:
ProjectionDefinition<'tokenUsage', TokenUsageState> = {
@@ -98,3 +111,41 @@ ProjectionDefinition<'tokenUsage', TokenUsageState> = {
view: state => state.totals,
stateVersion: 1,
}
/**
* Token-meter's context-occupancy projection unit.
*
* Two independent last-wins slots: the newest usage sample supplies the
* numerator, the newest `request/context` record the denominator. Both are
* whole values, so replay order alone decides the result and no cross-field
* consistency is claimed — the pair is explicitly not one atomic request
* observation (see {@link ContextPressureProjection}).
*
* The numerator is prompt-side only, so it holds still while a turn streams
* and steps forward once the next request reports its usage.
*/
export const contextPressureProjectionDefinition:
ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
key: 'contextPressure',
schema: pressureSchema,
init: () => ({ pressureTokens: 0 }),
apply: (state, event) => {
if (event.type === 'request/context') {
return event.data.contextWindow === state.contextWindow
? state
: { ...state, contextWindow: event.data.contextWindow }
}
const usage = event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
? event.data.chunk.usage
: event.type === 'assistant/message'
? event.data.usage
: undefined
if (usage === undefined) return state
const pressureTokens = pressureFrom(usage)
return pressureTokens === state.pressureTokens
? state
: { ...state, pressureTokens }
},
view: state => state,
stateVersion: 0,
}