Files
deepseek-harness/packages/llm/token-meter/src/usage-projection.ts
Hypatia May 8a8c1965d7 feat(web): show durable token usage and context occupancy in the stats line
The chat stats line took its token totals from the loaded conversation nodes,
so paging changed them and compaction erased the billing behind replaced
content. It also had no way to show context occupancy: the numerator and
capacity never reached the browser.

Both now come from token-meter session projections read through the standard
useProjection seat. Window nodes keep supplying turn and step counts plus LLM
and tool wall times, which are correctly window-scoped facts about what is on
screen; accounting no longer comes from there.

`tokenUsage` supplies billing and cache hit. `contextPressure` supplies
occupancy, pairing the newest provider-reported prompt size with the newest
capacity recorded by `request/context`. Deployments without token-meter drop
the token groups; a route whose adapter advertises no capacity drops the
occupancy group rather than rendering a placeholder.

Occupancy is deliberately approximate: the numerator and capacity are
independent last-wins fields, not one atomic request observation, so switching
models pairs a fresh capacity with the prior route's pressure until the next
request reports usage. It is a user-facing reference figure that nothing in the
harness makes decisions from, and it matches how the TUI status line has always
computed occupancy. The Agent Note and token-meter README state this as a
decision, including why the atomic alternative was implemented and rejected, so
it is not re-litigated as a defect.

Snapshot delta is one added `Context N% of 128K` segment across eight web
goldens; the preceding commit absorbed master's pre-existing golden drift.
2026-07-30 14:48:19 +08:00

152 lines
5.5 KiB
TypeScript

/**
* Pure folds for durable provider-reported token usage and context occupancy.
*/
import { z } from 'zod'
import type { TokenUsage } from '@deepseek-ai/dsh-llm'
import type { ProjectionDefinition } from '@deepseek-ai/dsh-session-projection'
import type { ContextPressureProjection, TokenUsageProjection } from './projection.ts'
interface UsageSample {
turn: number
step: number
buckets: TokenUsageProjection
}
interface TokenUsageState {
totals: TokenUsageProjection
last: UsageSample | null
}
const zeroBuckets = (): TokenUsageProjection => ({
uncachedInputTokens: 0,
outputTokens: 0,
cacheReadTokens: 0,
cacheWriteTokens: 0,
})
const bucketsFrom = (usage: TokenUsage): TokenUsageProjection => ({
uncachedInputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.cacheReadTokens ?? 0,
cacheWriteTokens: usage.cacheWriteTokens ?? 0,
})
const bucketsEqual = (left: TokenUsageProjection, right: TokenUsageProjection): boolean =>
left.uncachedInputTokens === right.uncachedInputTokens
&& left.outputTokens === right.outputTokens
&& left.cacheReadTokens === right.cacheReadTokens
&& left.cacheWriteTokens === right.cacheWriteTokens
const addReplacing = (
totals: TokenUsageProjection,
previous: TokenUsageProjection | undefined,
next: TokenUsageProjection,
): TokenUsageProjection => ({
uncachedInputTokens: totals.uncachedInputTokens - (previous?.uncachedInputTokens ?? 0) + next.uncachedInputTokens,
outputTokens: totals.outputTokens - (previous?.outputTokens ?? 0) + next.outputTokens,
cacheReadTokens: totals.cacheReadTokens - (previous?.cacheReadTokens ?? 0) + next.cacheReadTokens,
cacheWriteTokens: totals.cacheWriteTokens - (previous?.cacheWriteTokens ?? 0) + next.cacheWriteTokens,
})
const projectionSchema = z.object({
uncachedInputTokens: z.number().int().nonnegative(),
outputTokens: z.number().int().nonnegative(),
cacheReadTokens: z.number().int().nonnegative(),
cacheWriteTokens: z.number().int().nonnegative(),
}).strict()
// Cast for the optional capacity: under exactOptionalPropertyTypes zod infers
// `number | undefined` where the interface declares an absent-or-number field.
const pressureSchema = z.object({
pressureTokens: z.number().int().nonnegative(),
contextWindow: z.number().int().positive().optional(),
}).strict() as unknown as z.ZodType<ContextPressureProjection>
/** Prompt-side pressure of one request: input plus cache traffic, no output. */
const pressureFrom = (usage: TokenUsage): number =>
usage.inputTokens + (usage.cacheReadTokens ?? 0) + (usage.cacheWriteTokens ?? 0)
/**
* Token-meter's session projection unit.
*
* Usage chunks provide an early sample that survives a later request failure;
* an assistant message provides the final sample for the same turn/step. A
* repeated sample replaces that step's earlier value instead of double
* counting it. The single `last` slot relies on the session-log invariant
* that usage reports for one turn/step are adjacent: once a later step begins,
* a legal log never reports usage for an earlier step again.
*/
export const tokenUsageProjectionDefinition:
ProjectionDefinition<'tokenUsage', TokenUsageState> = {
key: 'tokenUsage',
schema: projectionSchema,
init: () => ({ totals: zeroBuckets(), last: null }),
apply: (state, event) => {
let turn: number
let step: number
let usage: TokenUsage
if (event.type === 'assistant/chunk' && event.data.chunk.type === 'usage') {
;({ turn, step } = event.data)
usage = event.data.chunk.usage
} else if (event.type === 'assistant/message' && event.data.usage !== undefined) {
;({ turn, step, usage } = event.data)
} else {
return state
}
const buckets = bucketsFrom(usage)
const previous = state.last !== null
&& state.last.turn === turn
&& state.last.step === step
? state.last.buckets
: undefined
if (previous !== undefined && bucketsEqual(previous, buckets)) return state
return {
totals: addReplacing(state.totals, previous, buckets),
last: { turn, step, buckets },
}
},
view: state => state.totals,
stateVersion: 1,
}
/**
* Token-meter's context-occupancy projection unit.
*
* Two independent last-wins slots: the newest usage sample supplies the
* numerator, the newest `request/context` record the denominator. Both are
* whole values, so replay order alone decides the result and no cross-field
* consistency is claimed — the pair is explicitly not one atomic request
* observation (see {@link ContextPressureProjection}).
*
* The numerator is prompt-side only, so it holds still while a turn streams
* and steps forward once the next request reports its usage.
*/
export const contextPressureProjectionDefinition:
ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
key: 'contextPressure',
schema: pressureSchema,
init: () => ({ pressureTokens: 0 }),
apply: (state, event) => {
if (event.type === 'request/context') {
return event.data.contextWindow === state.contextWindow
? state
: { ...state, contextWindow: event.data.contextWindow }
}
const usage = event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
? event.data.chunk.usage
: event.type === 'assistant/message'
? event.data.usage
: undefined
if (usage === undefined) return state
const pressureTokens = pressureFrom(usage)
return pressureTokens === state.pressureTokens
? state
: { ...state, pressureTokens }
},
view: state => state,
stateVersion: 1,
}