refactor(token-meter): make context occupancy durable projection state

Replace the transient `session/model-request` mux frame with ordinary durable
session state. Occupancy now rides two last-wins projection fields instead of a
non-replayable frame that needed removal tombstones and cross-stream fencing.

The frame was the only non-replayable class on the mux stream. Because host and
mux are independent SSE streams with no cross-stream order, a request emitted
before a removal could arrive after `host/session-removed`, and a legitimate
request for a new lifecycle reusing the same id could be fenced by a late
removal. Fixing that needed a lifecycle generation on every frame; the frame
itself was the problem.

Removed: the `session/model-request` frame and schema, the `agent/model-request`
core event, the ApiProxy measurement point, the client-side telemetry map and
removal tombstone, and the synthetic `cancelled` open error used to signal
reconnect through the error channel.

Added: `request/context`, a log-only session event recording the
registration-bound capacity of the route a request resolved to, appended beside
`request/header` from the lookup that already prepared the call and skipped when
the route is unchanged. Capacity stays out of `EpochHeader` because it is
adapter metadata about a route, not an input the request was built from, so it
must not join request reconstruction or header equality.

The `contextPressure` projection pairs the newest provider-reported prompt size
with the newest recorded capacity. The two are deliberately not one atomic
request observation: switching models can pair a fresh capacity with the prior
route's pressure until the next request reports usage. The figure is a
user-facing reference, and this matches how the TUI status line has always
computed occupancy.
This commit is contained in:
Hypatia May
2026-07-30 13:53:08 +08:00
parent fdccc58cef
commit 4819210142
62 changed files with 382 additions and 1362 deletions

View File

@@ -18,7 +18,7 @@ import type {
TokenMeterConfig,
TokenSurfaceNode,
} from './types.ts'
import { tokenUsageProjectionDefinition } from './usage-projection.ts'
import { contextPressureProjectionDefinition, tokenUsageProjectionDefinition } from './usage-projection.ts'
export type * from './types.ts'
@@ -97,6 +97,7 @@ export class TokenMeterService extends Service {
// compositions without the generic registry keep the meter's old shape.
ctx.inject(['sessionProjections'], (projectionCtx) => {
projectionCtx.sessionProjections.register(tokenUsageProjectionDefinition)
projectionCtx.sessionProjections.register(contextPressureProjectionDefinition)
})
// Readers catch up independently, while eager observation bounds ordinary

View File

@@ -1,5 +1,5 @@
/**
* Pure client-safe token-usage projection vocabulary.
* Pure client-safe token-projection vocabulary.
*
* @module @deepseek-ai/dsh-token-meter/projection
*/
@@ -17,9 +17,34 @@ export interface TokenUsageProjection {
cacheWriteTokens: number
}
/**
* Approximate context occupancy for a status display.
*
* The two fields are deliberately NOT one atomic request observation:
* `pressureTokens` is the newest provider-reported prompt size in the log,
* `contextWindow` the newest recorded route capacity. Switching models can
* therefore pair a fresh capacity with the previous route's pressure until the
* next request reports usage. This is an intentional trade — the value is a
* user-facing reference, not a billing or gating input — and it matches how
* the TUI status line has always computed occupancy. See the token-meter
* README for the full rationale.
*/
export interface ContextPressureProjection {
/**
* Provider-reported prompt size of the most recent request: uncached input
* plus cache reads and writes. Response output is excluded, so this does not
* grow as the current turn streams.
*/
pressureTokens: number
/** Newest recorded route capacity; absent when no adapter advertised one. */
contextWindow?: number
}
declare module '@deepseek-ai/dsh-session-projection/types' {
interface SessionProjectionMap {
/** Provider-reported usage accumulated across the complete durable log. */
tokenUsage: TokenUsageProjection
/** Newest request pressure paired with the newest known route capacity. */
contextPressure: ContextPressureProjection
}
}

View File

@@ -1,11 +1,11 @@
/**
* Pure fold for durable provider-reported token usage.
* Pure folds for durable provider-reported token usage and context occupancy.
*/
import { z } from 'zod'
import type { TokenUsage } from '@deepseek-ai/dsh-llm'
import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection'
import type { TokenUsageProjection } from './projection.ts'
import type { ContextPressureProjection, TokenUsageProjection } from './projection.ts'
interface UsageSample {
turn: number
@@ -56,13 +56,26 @@ const projectionSchema = z.object({
cacheWriteTokens: z.number().int().nonnegative(),
}).strict()
// Cast for the optional capacity: under exactOptionalPropertyTypes zod infers
// `number | undefined` where the interface declares an absent-or-number field.
const pressureSchema = z.object({
pressureTokens: z.number().int().nonnegative(),
contextWindow: z.number().int().positive().optional(),
}).strict() as unknown as z.ZodType<ContextPressureProjection>
/** Prompt-side pressure of one request: input plus cache traffic, no output. */
const pressureFrom = (usage: TokenUsage): number =>
usage.inputTokens + (usage.cacheReadTokens ?? 0) + (usage.cacheWriteTokens ?? 0)
/**
* Token-meter's session projection unit.
*
* Usage chunks provide an early sample that survives a later request failure;
* an assistant message provides the final sample for the same turn/step. A
* repeated sample replaces that step's earlier value instead of double
* counting it.
* counting it. The single `last` slot relies on the session-log invariant
* that usage reports for one turn/step are adjacent: once a later step begins,
* a legal log never reports usage for an earlier step again.
*/
export const tokenUsageProjectionDefinition:
ProjectionDefinition<'tokenUsage', TokenUsageState> = {
@@ -98,3 +111,41 @@ ProjectionDefinition<'tokenUsage', TokenUsageState> = {
view: state => state.totals,
stateVersion: 1,
}
/**
* Token-meter's context-occupancy projection unit.
*
* Two independent last-wins slots: the newest usage sample supplies the
* numerator, the newest `request/context` record the denominator. Both are
* whole values, so replay order alone decides the result and no cross-field
* consistency is claimed — the pair is explicitly not one atomic request
* observation (see {@link ContextPressureProjection}).
*
* The numerator is prompt-side only, so it holds still while a turn streams
* and steps forward once the next request reports its usage.
*/
export const contextPressureProjectionDefinition:
ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
key: 'contextPressure',
schema: pressureSchema,
init: () => ({ pressureTokens: 0 }),
apply: (state, event) => {
if (event.type === 'request/context') {
return event.data.contextWindow === state.contextWindow
? state
: { ...state, contextWindow: event.data.contextWindow }
}
const usage = event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
? event.data.chunk.usage
: event.type === 'assistant/message'
? event.data.usage
: undefined
if (usage === undefined) return state
const pressureTokens = pressureFrom(usage)
return pressureTokens === state.pressureTokens
? state
: { ...state, pressureTokens }
},
view: state => state,
stateVersion: 0,
}