fix(token-meter): close projected usage review gaps

This commit is contained in:
Hypatia May
2026-07-30 17:22:15 +08:00
parent e23cd8e406
commit 6d58953f30
43 changed files with 280 additions and 183 deletions

View File

@@ -475,6 +475,33 @@ interface FixtureTokenUsageProjection {
cacheWriteTokens: number
}
interface FixtureUsageSample {
turn: number
step: number
usage: TokenUsage
}
/** Read one provider usage sample from either durable carrier. */
function usageSampleOf(event: SessionEvent): FixtureUsageSample | undefined {
const item = event as unknown as {
type: string
data: {
turn?: number
step?: number
usage?: TokenUsage
chunk?: { type?: string; usage?: TokenUsage }
}
}
const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage'
? item.data.chunk.usage
: item.type === 'assistant/message'
? item.data.usage
: undefined
return usage === undefined || item.data.turn === undefined || item.data.step === undefined
? undefined
: { turn: item.data.turn, step: item.data.step, usage }
}
/** Fixture parallel of token-meter's last-sample-replacing usage projection. */
function tokenUsageOf(log: readonly SessionEvent[]): FixtureTokenUsageProjection {
const totals: FixtureTokenUsageProjection = {
@@ -489,47 +516,40 @@ function tokenUsageOf(log: readonly SessionEvent[]): FixtureTokenUsageProjection
buckets: FixtureTokenUsageProjection
} | null = null
for (const event of log) {
const item = event as unknown as {
type: string
data: {
turn?: number
step?: number
usage?: TokenUsage
chunk?: { type?: string; usage?: TokenUsage }
}
}
const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage'
? item.data.chunk.usage
: item.type === 'assistant/message'
? item.data.usage
: undefined
if (usage === undefined || item.data.turn === undefined || item.data.step === undefined) continue
const sample = usageSampleOf(event)
if (sample === undefined) continue
const buckets: FixtureTokenUsageProjection = {
uncachedInputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.cacheReadTokens ?? 0,
cacheWriteTokens: usage.cacheWriteTokens ?? 0,
uncachedInputTokens: sample.usage.inputTokens,
outputTokens: sample.usage.outputTokens,
cacheReadTokens: sample.usage.cacheReadTokens ?? 0,
cacheWriteTokens: sample.usage.cacheWriteTokens ?? 0,
}
const previous = last?.turn === item.data.turn && last.step === item.data.step
const previous = last?.turn === sample.turn && last.step === sample.step
? last.buckets
: undefined
totals.uncachedInputTokens += buckets.uncachedInputTokens - (previous?.uncachedInputTokens ?? 0)
totals.outputTokens += buckets.outputTokens - (previous?.outputTokens ?? 0)
totals.cacheReadTokens += buckets.cacheReadTokens - (previous?.cacheReadTokens ?? 0)
totals.cacheWriteTokens += buckets.cacheWriteTokens - (previous?.cacheWriteTokens ?? 0)
last = { turn: item.data.turn, step: item.data.step, buckets }
last = { turn: sample.turn, step: sample.step, buckets }
}
return totals
}
/** Latest log-only capacity record, or undefined before any request ran. */
interface FixtureRequestContext {
provider: string
model: string
contextWindow?: number
}
/** Latest log-only route context, or undefined before any request ran. */
function lastRequestContext(
log: readonly SessionEvent[],
): { provider: string; model: string; contextWindow: number } | undefined {
): FixtureRequestContext | undefined {
const event = log.findLast(item => (item as { type: string }).type === 'request/context')
return event === undefined
? undefined
: (event as unknown as { data: { provider: string; model: string; contextWindow: number } }).data
: (event as unknown as { data: FixtureRequestContext }).data
}
/**
@@ -539,26 +559,18 @@ function lastRequestContext(
*/
function contextPressureOf(
log: readonly SessionEvent[],
): { pressureTokens: number; contextWindow?: number } {
let pressureTokens = 0
): { pressureTokens?: number; contextWindow?: number } {
let pressureTokens: number | undefined
for (const event of log) {
const item = event as unknown as {
type: string
data: { usage?: TokenUsage; chunk?: { type?: string; usage?: TokenUsage } }
}
const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage'
? item.data.chunk.usage
: item.type === 'assistant/message'
? item.data.usage
: undefined
if (usage === undefined) continue
pressureTokens = usage.inputTokens
+ (usage.cacheReadTokens ?? 0)
+ (usage.cacheWriteTokens ?? 0)
const sample = usageSampleOf(event)
if (sample === undefined) continue
pressureTokens = sample.usage.inputTokens
+ (sample.usage.cacheReadTokens ?? 0)
+ (sample.usage.cacheWriteTokens ?? 0)
}
const contextWindow = lastRequestContext(log)?.contextWindow
return {
pressureTokens,
...pressureTokens === undefined ? {} : { pressureTokens },
...contextWindow === undefined ? {} : { contextWindow },
}
}
@@ -588,12 +600,7 @@ function projectionValuesOf(log: readonly SessionEvent[]): Record<string, unknow
function projectionFramesOf(id: SessionId, log: readonly SessionEvent[], event: SessionEvent): Extract<MuxFrame, { type: 'session/projection' }>[] {
const type = (event as { type: string }).type
// One usage sample advances both token-meter units.
if (
(type === 'assistant/chunk'
&& (event as unknown as { data: { chunk?: { type?: string } } }).data.chunk?.type === 'usage')
|| (type === 'assistant/message'
&& (event as unknown as { data: { usage?: TokenUsage } }).data.usage !== undefined)
) {
if (usageSampleOf(event) !== undefined) {
return [
{ type: 'session/projection', sessionId: id, key: 'tokenUsage', value: tokenUsageOf(log), seq: event.seq },
{ type: 'session/projection', sessionId: id, key: 'contextPressure', value: contextPressureOf(log), seq: event.seq },

View File

@@ -90,8 +90,8 @@ describe('createFixtureApi', () => {
cacheReadTokens: 0,
cacheWriteTokens: 0,
},
// No request ran, so pressure is zero and no capacity is known yet.
contextPressure: { pressureTokens: 0 },
// No request ran, so neither pressure nor capacity is known yet.
contextPressure: {},
} },
})
})

View File

@@ -2,5 +2,5 @@
# side as of the last confirmed-consistent state. Both languages carry equal authority;
# after editing either side, bring the other along and re-record with:
# pnpm run verify-translation-pairing --write packages/client/ui-conversation/README.md
README.md: 84051dbe29503bcb20e317247db2ba2026db4528
README.zh.md: 36a0983f95dafc55a4fc76a5bc82112fb369efce
README.md: c5b10490abf42bcb04a675d50b3c487fdbf3f3b8
README.zh.md: e8bca83ae56c638e7a8fb852c35164b1941d749f

View File

@@ -22,7 +22,7 @@ Per-session UI state for selection and the active view lives in the declared cha
The composer bar declares session-scoped single seats for `'conversation.input.plan'` (right of the local access-mode control) and `'conversation.input.model'` (immediately before the pending indicator and send/stop button), plus list slots for overlay, dock, left, and right input extensions. Feature packages own each control and its state; ui-conversation supplies placement, the `locked` owner prop, and the standard slot shares. While the `plan` projection's effective target is plan mode, InputBar swaps its textarea placeholder to the plan-task wording, localized through the `command.hint` locale namespace this package registers and shared verbatim with the claimed `/plan` command hint (a host-folded value read through the standard-kit `useProjection`; owner-supplied placeholders win). A pending composer takeover remains mounted when another conversation view is active so the blocked agent can still receive its answer; without a pending interaction, the active-session composer belongs to Chat. The resident no-session shell uses `DisabledInputBar` and therefore dispatches no session-scoped control seats.
The chat stats line takes its token accounting from two generic token-meter projections read through the standard-kit `useProjection`: `tokenUsage` for full-log billing (cache hit is `cacheRead / (uncachedInput + cacheRead)`, excluding cache writes) and `contextPressure` for context occupancy. Visible nodes supply only the turn and step counts plus the LLM and tool wall times, which are window-scoped facts about what is on screen rather than accounting. A deployment without token-meter drops the token groups, and a route whose adapter advertises no capacity drops the occupancy group instead of rendering a placeholder. Occupancy is deliberately an approximation — its numerator and capacity are independent last-wins projection fields, not one atomic request observation ([rationale](../../llm/token-meter/README.md)). The inline stats row remains the sole context UI; the model selector has no circle or accessory.
The chat stats line takes its token accounting from two generic token-meter projections read through the standard-kit `useProjection`: `tokenUsage` for full-log billing (billed input is uncached input plus cache reads and writes; cache hit divides cache reads by that total) and `contextPressure` for context occupancy. Visible nodes supply only the turn and step counts plus the LLM and tool wall times, which are window-scoped facts about what is on screen rather than accounting; durable token and context groups remain visible when compaction leaves no assistant node in the loaded window. A deployment without token-meter drops the token groups, and occupancy stays hidden until both provider pressure and route capacity are known. Occupancy is deliberately an approximation — its numerator and capacity are independent last-wins projection fields, not one atomic request observation ([rationale](../../llm/token-meter/README.md)). The inline stats row remains the sole context UI; the model selector has no circle or accessory.
`src/client/` is organized for the future package split: `contract/` is the sole inter-domain shared face (`slots.ts` slot declarations + composed slot props including the tool-row contract, `views.ts` shared primitives, `tool-call-model.ts`); the `skeleton/`, `chat/`, and `toolviews/` (sample registrants) domain directories import contract files and never each other; `apply.ts` is the only assembly point allowed to import all three domains. The `/client` export surface is the contract only — `apply`/`inject`, the two service classes, and the `contract/` type families; implementation components (skeleton, chat rows) and the store factory stay internal and reach the page exclusively through apply's slot registrations (tests take them via the `./src/*` subpath).

View File

@@ -22,7 +22,7 @@ todo 两个面就是在该形状上的两个注册项,都是普通注册方插
输入栏为 `'conversation.input.plan'`(位于本地 access 模式控件右侧)和 `'conversation.input.model'`(渲染在 pending 指示器与发送/停止按钮之前)声明会话作用域的单实例 seat,并为 overlay、dock、left 和 right 输入扩展声明列表 slot。各功能包拥有相应控件及其状态;ui-conversation 提供放置位置、`locked` owner prop 和标准 slot share。当 `plan` 投影的有效目标为 plan mode 时,InputBar 将文本框 placeholder 切换为 plan 任务措辞,经本包注册的 `command.hint` locale 命名空间本地化,并与已认领 `/plan` 命令的提示逐字共用同一份文案(经标准套件 `useProjection` 读取的 host 折叠值;owner 提供的 placeholder 优先)。另一个会话视图活跃时,待处理的 composer 接管仍保持挂载,使被阻塞的 agent(智能体)仍能收到回答;没有待处理交互时,活跃会话的 composer 归 Chat 所有。常驻无会话壳使用 `DisabledInputBar`,因此不会分发任何会话作用域的控件 seat。
聊天统计行的 token 账目来自经标准套件 `useProjection` 读取的两个通用 token-meter 投影:`tokenUsage` 提供完整日志计费用量(缓存命中率为 `cacheRead / (uncachedInput + cacheRead)`,不计入缓存写入),`contextPressure` 提供上下文占用率。可见节点只提供轮次与步骤计数,以及 LLM 和工具的墙钟时间:这些是关于「屏幕上有什么」的窗口作用域事实,而非账目。未组合 token-meter 的部署会整组省略 token 分组;适配器未公布容量的路由会省略占用率分组,而不是渲染占位文案。占用率是刻意为之的近似值:它的分子与容量是两个相互独立的「后者胜」投影字段,并非同一次请求的原子观测([原理](../../llm/token-meter/README.md))。行内统计行仍是唯一的上下文 UI;模型选择器不增加圆环或附属控件。
聊天统计行的 token 账目来自经标准套件 `useProjection` 读取的两个通用 token-meter 投影:`tokenUsage` 提供完整日志计费用量(计费输入为未缓存输入、缓存读取与缓存写入之和;缓存命中率以缓存读取除以该总量),`contextPressure` 提供上下文占用率。可见节点只提供轮次与步骤计数,以及 LLM 和工具的墙钟时间:这些是关于「屏幕上有什么」的窗口作用域事实,而非账目;压缩使已加载窗口不再包含 assistant 节点时,持久 token 与上下文分组仍保持可见。未组合 token-meter 的部署会整组省略 token 分组;只有提供方压力与路由容量都已知时才显示占用率。占用率是刻意为之的近似值:它的分子与容量是两个相互独立的「后者胜」投影字段,并非同一次请求的原子观测([原理](../../llm/token-meter/README.md))。行内统计行仍是唯一的上下文 UI;模型选择器不增加圆环或附属控件。
`src/client/` 按未来的包拆分组织:`contract/` 是唯一的跨领域共享表层(`slots.ts` slot 声明 + 组合后的 slot props,包括工具行契约、`views.ts` 共享原语、`tool-call-model.ts`);`skeleton/`、`chat/` 和 `toolviews/`(示例注册方)领域目录只导入 contract 文件,彼此绝不导入;`apply.ts` 是唯一允许导入全部三个领域的组装点。`/client` 导出表层只包含契约:`apply`/`inject`、两个服务类和 `contract/` 类型家族;实现组件(骨架、聊天行)与 store factory 保持内部状态,只能通过 apply 的 slot 注册到达页面(测试通过 `./src/*` 子路径获取它们)。

View File

@@ -78,23 +78,38 @@ export function formatDuration(ms: number): string {
* @returns rounded integer percent, or null when no input was billed.
*/
export function cacheHitPercent(usage: TokenUsageProjection): number | null {
const denominator = usage.uncachedInputTokens + usage.cacheReadTokens
const denominator = billedInputTokens(usage)
return denominator === 0
? null
: Math.round(usage.cacheReadTokens / denominator * 100)
}
/** Sum the three disjoint prompt-side billing buckets. */
function billedInputTokens(usage: TokenUsageProjection): number {
return usage.uncachedInputTokens + usage.cacheReadTokens + usage.cacheWriteTokens
}
interface ContextOccupancy {
percent: number
contextWindow: number
}
/**
* Approximate context occupancy, using the TUI's integer rounding and upper
* clamp. The numerator and capacity are independent last-wins projection
* fields, so this is a reference figure rather than an exact measurement of one
* request (see the token-meter README).
* @param pressure - the session's context-pressure projection value.
* @returns occupancy percent, or null when no capacity is known.
* @returns occupancy and its denominator, or null until both values are known.
*/
export function contextPercent(pressure: ContextPressureProjection | undefined): number | null {
if (pressure?.contextWindow === undefined) return null
return Math.min(100, Math.round(pressure.pressureTokens / pressure.contextWindow * 100))
export function contextOccupancy(
pressure: ContextPressureProjection | undefined,
): ContextOccupancy | null {
if (pressure?.pressureTokens === undefined || pressure.contextWindow === undefined) return null
return {
percent: Math.min(100, Math.round(pressure.pressureTokens / pressure.contextWindow * 100)),
contextWindow: pressure.contextWindow,
}
}
/** Props: the conversation-snapshot selector plus the projection read seat. */
@@ -108,29 +123,31 @@ export const StatsLine = memo(function StatsLine({ useSession, useProjection }:
const usage = useProjection('tokenUsage')
const pressure = useProjection('contextPressure')
const stats = useMemo(() => deriveStats(nodes), [nodes])
if (stats.steps === 0) return null
// Pipe-separated groups (figma stats strip); a group with no data drops out whole.
const groups: string[] = [`${stats.turns} turns · ${stats.steps} steps`]
const durations: string[] = []
if (stats.llmMs > 0) durations.push(`LLM ${formatDuration(stats.llmMs)}`)
if (stats.toolMs > 0) durations.push(`Tool call ${formatDuration(stats.toolMs)}`)
if (durations.length > 0) groups.push(durations.join(' · '))
const context = contextPercent(pressure)
// Capacity absent (no token-meter composed, or an adapter that advertises
// none) drops the group: an unknown denominator has no percentage to show.
if (context !== null && pressure?.contextWindow !== undefined) {
groups.push(`Context ${context}% of ${formatTokens(pressure.contextWindow)}`)
const groups: string[] = []
if (stats.steps > 0) {
groups.push(`${stats.turns} turns · ${stats.steps} steps`)
const durations: string[] = []
if (stats.llmMs > 0) durations.push(`LLM ${formatDuration(stats.llmMs)}`)
if (stats.toolMs > 0) durations.push(`Tool call ${formatDuration(stats.toolMs)}`)
if (durations.length > 0) groups.push(durations.join(' · '))
}
const context = contextOccupancy(pressure)
if (context !== null) {
groups.push(`Context ${context.percent}% of ${formatTokens(context.contextWindow)}`)
}
// Billing rides the durable projection, so these survive paging and
// compaction; a deployment without token-meter drops the groups entirely.
if (usage !== undefined) {
// compaction. Suppress the empty projection on a brand-new session.
if (usage !== undefined
&& (stats.steps > 0 || billedInputTokens(usage) > 0 || usage.outputTokens > 0)) {
const cacheHit = cacheHitPercent(usage)
if (cacheHit !== null) groups.push(`Cache hit ${cacheHit}%`)
groups.push(
`Input ${formatTokens(usage.uncachedInputTokens + usage.cacheReadTokens)} tok`
`Input ${formatTokens(billedInputTokens(usage))} tok`
+ ` · Output ${formatTokens(usage.outputTokens)} tok`,
)
}
if (groups.length === 0) return null
return (
<div className={css.root}>
{groups.map((group, i) => (

View File

@@ -218,8 +218,8 @@ describe('small branch tails', () => {
})
it('StatsLine omits the cache-hit segment when no input accounting exists at all', () => {
// Cache hit is null only when uncached input and cache reads are both zero
// (pure output accounting) — any input makes it a real 0%.
// Cache hit is null only when all three prompt buckets are zero (pure
// output accounting) — any billed input makes it a real 0%.
const snap = {
nodes: [{ kind: 'assistant', seq: 1, turn: 1, step: 1, blocks: [], usage: { outputTokens: 10 } }],
}

View File

@@ -122,17 +122,30 @@ describe('StatsLine', () => {
return { useSession: bindSnapshotSelector(source), useProjection: projections(values) }
}
it('renders the grouped stats row and hides with zero steps', () => {
it('renders the grouped stats row and hides a brand-new empty session', () => {
const { source } = makeSource({ nodes: [assistant(1, 1)] })
const view = render(<StatsLine {...props(source)} />)
// No timing on the fixture: the duration group drops out whole. Tokens come
// from the projection, so paging the window cannot change them.
expect(view.container.textContent).toBe('1 turns · 1 steps|Cache hit 90%|Input 100 tok · Output 5 tok')
const empty = makeSource()
const emptyView = render(<StatsLine {...props(empty.source)} />)
const emptyView = render(<StatsLine {...props(empty.source, {
tokenUsage: { uncachedInputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0 },
contextPressure: {},
})} />)
expect(emptyView.container.textContent).toBe('')
})
it('keeps durable token and context groups after the visible step window is empty', () => {
const { source } = makeSource()
const view = render(<StatsLine {...props(source, {
tokenUsage: USAGE,
contextPressure: { pressureTokens: 32_000, contextWindow: 128_000 },
})} />)
expect(view.container.textContent)
.toBe('Context 25% of 128K|Cache hit 90%|Input 100 tok · Output 5 tok')
})
it('renders context occupancy only when the projection knows a capacity', () => {
const { source } = makeSource({ nodes: [assistant(1, 1)] })
const withCapacity = render(<StatsLine {...props(source, {
@@ -146,6 +159,13 @@ describe('StatsLine', () => {
contextPressure: { pressureTokens: 32_000 },
})} />)
expect(noCapacity.container.textContent).not.toContain('Context')
// Capacity arrives before usage in the log; no provider sample means there
// is no numerator yet, rather than a synthetic 0%.
const noPressure = render(<StatsLine {...props(source, {
tokenUsage: USAGE,
contextPressure: { contextWindow: 128_000 },
})} />)
expect(noPressure.container.textContent).not.toContain('Context')
})
it('clamps occupancy at 100% when pressure exceeds the recorded capacity', () => {
@@ -173,6 +193,20 @@ describe('StatsLine', () => {
expect(view.container.textContent).toBe('1 turns · 1 steps|Input 0 tok · Output 7 tok')
})
it('includes cache writes in billed input and the cache-hit denominator', () => {
const { source } = makeSource({ nodes: [assistant(1, 1)] })
const view = render(<StatsLine {...props(source, {
tokenUsage: {
uncachedInputTokens: 10,
outputTokens: 7,
cacheReadTokens: 90,
cacheWriteTokens: 100,
},
})} />)
expect(view.container.textContent)
.toBe('1 turns · 1 steps|Cache hit 45%|Input 200 tok · Output 7 tok')
})
it('renders ZERO times during streaming chunk frames (RFC hard acceptance)', () => {
const { set, source } = makeSource({ nodes: [assistant(1, 1)] })
let renders = 0

View File

@@ -2105,7 +2105,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [
},
{
name: 'RequestContext',
declaration: 'export interface RequestContext {\n provider: string;\n model: string;\n contextWindow: number;\n}',
declaration: 'export interface RequestContext {\n provider: string;\n model: string;\n contextWindow?: number;\n}',
},
{
name: 'RequestHeaderReason',

View File

@@ -44,7 +44,7 @@ import {
} from '@deepseek-ai/dsh-llm'
import type { GenerateOptions, LlmCallConfig, LlmFailure, Message, PreparedLlmCall, ResolvedRetryPolicy } from '@deepseek-ai/dsh-llm'
import { canonicalHeader, headerEquals } from '@deepseek-ai/dsh-session'
import type { AssistantMessage, Session, SessionId, TurnEndReason, TurnTrigger, UserMessage } from '@deepseek-ai/dsh-session'
import type { AssistantMessage, RequestContext, Session, SessionId, TurnEndReason, TurnTrigger, UserMessage } from '@deepseek-ai/dsh-session'
import { renderPrompt } from '@deepseek-ai/dsh-system-prompt'
import type {} from '@deepseek-ai/dsh-tools'
import { executeToolCalls } from './tool-calls.ts'
@@ -668,21 +668,21 @@ export class ReactLoopAgent implements Agent {
session.append('request/header', { header, reason: 'change' })
}
// Capacity of the route this request resolved to, recorded from the same
// Context metadata for the route this request resolved to, recorded from the same
// registration-bound lookup that prepared the call (no second resolve).
// Deduplicated against the last record: an unchanged route logs nothing.
// A route with unknown capacity is still recorded so it clears any older
// denominator; an unchanged route logs nothing.
const contextWindow = preparedCall?.context?.contextWindow
if (contextWindow !== undefined) {
const previous = session.requestContext()
if (previous?.provider !== config.provider
|| previous.model !== config.model
|| previous.contextWindow !== contextWindow) {
session.append('request/context', {
provider: config.provider,
model: config.model,
contextWindow,
})
}
const requestContext: RequestContext = {
provider: config.provider,
model: config.model,
...contextWindow === undefined ? {} : { contextWindow },
}
const previous = session.requestContext()
if (previous?.provider !== requestContext.provider
|| previous.model !== requestContext.model
|| previous.contextWindow !== requestContext.contextWindow) {
session.append('request/context', requestContext)
}
const request = markAgentLoopRequest(deepFreeze({

View File

@@ -578,13 +578,38 @@ describe('request/context capacity records', () => {
.map(event => event.data.contextWindow)).toEqual([64_000, 256_000])
})
it('records nothing when the adapter advertises no capacity', async () => {
// The absent-capacity path must stay silent rather than log a placeholder:
// consumers read "no capacity known" and omit their percentage entirely.
const ctx = await harness(new MockAdapter([textResponse('a')]))
it('records and deduplicates a route whose adapter advertises no capacity', async () => {
const ctx = await harness(new MockAdapter([textResponse('a'), textResponse('b')]))
const agent = ctx.agentLoop.create(SessionId('capacity-absent'), { provider: 'mock', model: 'mock' })
send(agent, 'go')
send(agent, 'first')
await waitForIdle(ctx, agent)
expect(agent.session.events.some(event => event.type === 'request/context')).toBe(false)
send(agent, 'second')
await waitForIdle(ctx, agent)
expect(agent.session.events
.filter(event => event.type === 'request/context')
.map(event => event.data)).toEqual([{ provider: 'mock', model: 'mock' }])
})
it('clears a previous capacity when the next route advertises none', async () => {
const adapter = capacityAdapter({ known: 64_000 }, [textResponse('a'), textResponse('b')])
const ctx = await harness(adapter)
const agent = ctx.agentLoop.create(SessionId('capacity-clear'), { provider: 'mock', model: 'known' })
let model = 'known'
ctx.on('agent/request', (subject, _turn, _step, _signal, next) => subject === agent
? Promise.resolve({ provider: 'mock', model })
: next())
send(agent, 'first')
await waitForIdle(ctx, agent)
model = 'unknown'
send(agent, 'second')
await waitForIdle(ctx, agent)
expect(agent.session.events
.filter(event => event.type === 'request/context')
.map(event => event.data)).toEqual([
{ provider: 'mock', model: 'known', contextWindow: 64_000 },
{ provider: 'mock', model: 'unknown' },
])
})
})

View File

@@ -2,5 +2,5 @@
# side as of the last confirmed-consistent state. Both languages carry equal authority;
# after editing either side, bring the other along and re-record with:
# pnpm run verify-translation-pairing --write packages/core/session/README.md
README.md: 861b96c453e677807bded2475fe8e62a74bcd299
README.zh.md: 6974a072f5cb32f4e850846bbb02af59cda93303
README.md: fe96b5c9735d48d4f92210970b7707749a920787
README.zh.md: 6f8aaeef2464a946e10aeef0cfe13b36ab303aeb

View File

@@ -65,7 +65,7 @@ Providers stream token-sized deltas, so a raw log stores hundreds of `assistant/
`request/header` records a full canonical snapshot of the non-history request envelope with reason `initial`, `resume`, or `change`. `foldRequestHeader()` selects the latest snapshot; legacy delta events and the removed `fallback` reason are rejected. See the [reconstructable-requests Agent Note](../../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md).
`request/context` records the registration-bound `contextWindow` of the route a request resolved to, appended inside its step beside `request/header` and only when the provider, model, or capacity differs from the previous record. `session.requestContext()` folds the latest one incrementally, mirroring `requestHeader()`. Capacity stays OUT of `EpochHeader` on purpose: it is adapter metadata describing a route, not an input the request was built from, so it must not enter request reconstruction or header equality — a capacity change is not a header `change`. A route whose adapter advertises no capacity appends nothing.
`request/context` records registration-bound metadata for the route a request resolved to, appended inside its step beside `request/header` and only when the provider, model, or capacity differs from the previous record. `session.requestContext()` folds the latest one incrementally, mirroring `requestHeader()`. Capacity stays OUT of `EpochHeader` on purpose: it is adapter metadata describing a route, not an input the request was built from, so it must not enter request reconstruction or header equality — a capacity change is not a header `change`. A route whose adapter advertises no capacity is still recorded with `contextWindow` absent, clearing any older known capacity.
A `user/message` stores the complete `UserMessage` directly, including the identity created before routing or prompt admission. It renders its `content` verbatim whether it is a direct human prompt, a synthetic injection, or an admitted goal round; its typed `source` is the only channel that tells them apart and carries any domain-specific durable facts. `assistant/message`, `tool/result`, and `steering/message` likewise store complete message values. Turn execution remains enclosed by `turn/start` and `turn/end`, while an idle injection may append and flush a `user/message` between turns without running the model.

View File

@@ -65,7 +65,7 @@
`request/header` 记录非历史请求封装的完整规范快照,其原因为 `initial`、`resume` 或 `change`。`foldRequestHeader()` 选择最新快照;旧版增量事件和已移除的 `fallback` 原因会被拒绝。详见[可重建请求 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md)。
`request/context` 记录请求所解析到的路由的、绑定注册项的 `contextWindow`,在其所属步骤内紧随 `request/header` 追加,且仅在提供方、模型或容量与上一条记录不同时追加。`session.requestContext()` 以增量方式归并最新一条,与 `requestHeader()` 保持一致。容量刻意不进入 `EpochHeader`:它是描述路由的适配器元数据,不是构建该请求所依据的输入,因此绝不可进入请求重建或请求头相等性判断:容量变化不构成请求头 `change`。适配器不公布容量的路由不追加任何记录。
`request/context` 记录请求所解析到的路由的、绑定注册项的元数据,在其所属步骤内紧随 `request/header` 追加,且仅在提供方、模型或容量与上一条记录不同时追加。`session.requestContext()` 以增量方式归并最新一条,与 `requestHeader()` 保持一致。容量刻意不进入 `EpochHeader`:它是描述路由的适配器元数据,不是构建该请求所依据的输入,因此绝不可进入请求重建或请求头相等性判断:容量变化不构成请求头 `change`。适配器不公布容量的路由仍会被记录,但 `contextWindow` 字段缺失,从而清除较早的已知容量。
`user/message` 会直接存储完整的 `UserMessage`,其中包括路由或提示词准入前创建的标识。无论它是直接人类提示词、合成注入,还是已准入的 Goal Round,都会原样呈现其 `content`;带类型的 `source` 是区分三者的唯一通道,并携带各领域专有的持久事实。`assistant/message`、`tool/result` 和 steering(中途引导)对应的 `steering/message` 也会存储完整的消息值。轮次执行仍由 `turn/start` 与 `turn/end` 包围,而空闲注入可以在轮次之间追加并刷新一条 `user/message`,无需运行模型。

View File

@@ -584,11 +584,11 @@ export class Session {
private contextFoldSeq = 0
/**
* The route capacity in force after the log's last `request/context` event —
* The route metadata in force after the log's last `request/context` event —
* what the NEXT request deduplicates against — or undefined before any such
* record. Maintained incrementally like {@link requestHeader}, so a per-step
* read costs O(new events).
* @returns the folded capacity record, or undefined when none exists yet.
* @returns the folded context record, or undefined when none exists yet.
*/
requestContext(): RequestContext | undefined {
if (this.contextFoldSeq < this.log.length) {

View File

@@ -170,17 +170,17 @@ export interface EpochHeader {
}
/**
* Registration-bound context capacity of one resolved model route. Adapter
* Registration-bound context metadata of one resolved model route. Adapter
* metadata about a route rather than a request input, which is why it lives
* outside {@link EpochHeader}.
*/
export interface RequestContext {
/** Registered provider route the capacity was resolved through. */
/** Registered provider route the metadata was resolved through. */
provider: string
/** Provider-owned model id the capacity belongs to. */
/** Provider-owned model id the metadata belongs to. */
model: string
/** Maximum combined request and response context in tokens. */
contextWindow: number
/** Maximum combined request and response context in tokens; absent when the adapter advertises none. */
contextWindow?: number
}
/**
@@ -265,13 +265,13 @@ export interface SessionEventMap {
*/
'request/header': { header: EpochHeader; reason: RequestHeaderReason }
/**
* Registration-bound context capacity for the route a request resolved to,
* Registration-bound context metadata for the route a request resolved to,
* appended inside its step beside `request/header` and only when the route
* or capacity differs from the last record. It is log-only and deliberately
* NOT part of {@link EpochHeader}: capacity is adapter metadata about a
* route, not an input the request was built from, so it must not participate
* in request reconstruction or header equality. Absent for a route whose
* adapter advertises no capacity.
* in request reconstruction or header equality. `contextWindow` is absent
* when the route's adapter advertises no capacity.
*/
'request/context': RequestContext
/**

View File

@@ -96,7 +96,7 @@ describe('Session.requestContext', () => {
const CAPACITY = { provider: 'mock', model: 'm', contextWindow: 128_000 }
/** A turn-enclosed capacity record; the invariant rejects one outside a turn. */
function seedWith(...records: { provider: string; model: string; contextWindow: number }[]): SessionEvent[] {
function seedWith(...records: { provider: string; model: string; contextWindow?: number }[]): SessionEvent[] {
const events: SessionEvent[] = [{
type: 'turn/start', seq: 0, time: 1, data: { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } },
}]
@@ -127,6 +127,8 @@ describe('Session.requestContext', () => {
expect(session.requestContext()).toEqual(CAPACITY)
session.append('request/context', { ...CAPACITY, model: 'next', contextWindow: 64_000 })
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'next', contextWindow: 64_000 })
session.append('request/context', { provider: 'mock', model: 'unknown' })
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'unknown' })
})
it('folds a batch appended between two reads', () => {
@@ -143,6 +145,6 @@ describe('Session.requestContext', () => {
const held = session.requestContext()
if (held === undefined) throw new Error('expected a folded capacity record')
expect(Object.isFrozen(held)).toBe(true)
expect(() => { (held as { contextWindow: number }).contextWindow = 1 }).toThrow()
expect(() => { (held as { contextWindow?: number }).contextWindow = 1 }).toThrow()
})
})

View File

@@ -2,5 +2,5 @@
# side as of the last confirmed-consistent state. Both languages carry equal authority;
# after editing either side, bring the other along and re-record with:
# pnpm run verify-translation-pairing --write packages/llm/token-meter/README.md
README.md: b9bf1dfa253e424e5ec35cd3e7bf0f52af579077
README.zh.md: c97fc0b87364dfe9ca46139f0ec82519e191b772
README.md: 701893b342f9a93a75bec175634b1054f3d17151
README.zh.md: a5844e8788422bba669632ed587fb87e1e2a1e58

View File

@@ -27,7 +27,7 @@ When the composition provides `ctx.sessionProjections`, token-meter registers tw
`tokenUsage` carries the complete durable log's `uncachedInputTokens`, `outputTokens`, `cacheReadTokens`, and `cacheWriteTokens`. Usage chunks are counted even when a request later fails; a final assistant-message usage for the same `(turn, step)` replaces that sample instead of double-counting it. Reasoning remains an output subdivision. The single last-sample slot relies on a session-log ordering property: once a later step reports usage, a legal log never reports usage for an earlier step again.
`contextPressure` carries `pressureTokens` — the newest provider-reported prompt size, summing uncached input plus cache reads and writes — and the optional `contextWindow` from the newest `request/context` record. Output is excluded, so the numerator holds still while a turn streams and steps forward when the next request reports its usage.
`contextPressure` carries optional `pressureTokens` — the newest provider-reported prompt size, summing uncached input plus cache reads and writes — and optional `contextWindow` from the newest `request/context` record. Pressure stays absent until a provider reports usage; capacity stays absent for a route whose adapter advertises none. Output is excluded, so the numerator holds still while a turn streams and steps forward when the next request reports its usage.
Both units use the standard projection baseline, live frame, higher-seq-wins store, and JSON checkpoint paths. Unloading token-meter removes both keys. A headless or TUI composition without the projection seam keeps the measurement service's existing behavior.
@@ -62,3 +62,4 @@ No direct invalidation; the named consumer owns any request-prefix changes.
- **Every measurement clones the current surface** — coherent immutable snapshots make reads O(surface), including below-threshold pressure checks.
- **Provider usage is only reusable for an identical canonical envelope** — prompt, prefix, tools, provider, model, or call-config changes deliberately fall back to full heuristic estimation.
- **Legacy provenance is conservative** — assistant messages without `sourceEventSeqs` cannot distinguish provider output from listener rewrites, so the fold avoids claiming a known empty or exact chunk stream.
- **The TUI and browser fixture retain parallel folds** — `tokenUsage` owns durable session-projection semantics; the TUI keeps its live per-step map because its composition does not mount the generic projection seam, while the browser fixture mirrors the unit for standalone demo data.

View File

@@ -27,7 +27,7 @@ fold 跟踪完整请求标头快照、步骤边界、表层追加与替换、成
`tokenUsage` 携带完整持久日志中的 `uncachedInputTokens`、`outputTokens`、`cacheReadTokens` 和 `cacheWriteTokens`。即使请求随后失败,用量分片仍会计入;同一 `(turn, step)` 的最终 assistant 消息用量会替换该样本,而不是重复计数。推理仍是输出的一个细分项。只保留单个最新样本,依赖的是会话日志的一条顺序性质:一旦某个更晚的步骤报告了用量,合法日志就绝不会再为更早的步骤报告用量。
`contextPressure` 携带 `pressureTokens`(提供方报告的最新提示词规模,为未缓存输入加缓存读取与写入之和),以及来自最新一条 `request/context` 记录的可选 `contextWindow`。输出不计入其中,因此轮次流式输出期间分子保持不动,等到下一个请求报告用量时才前进。
`contextPressure` 携带可选的 `pressureTokens`(提供方报告的最新提示词规模,为未缓存输入加缓存读取与写入之和),以及来自最新一条 `request/context` 记录的可选 `contextWindow`。提供方报告用量前压力保持缺失;路由适配器未公布容量时容量也保持缺失。输出不计入其中,因此轮次流式输出期间分子保持不动,等到下一个请求报告用量时才前进。
两个单元都使用标准的投影基线、实时帧、seq 高者胜值仓和 JSON 检查点路径。卸载 token-meter 会移除这两个键。不带投影 seam 的 headless 或 TUI 组合会保留测量服务的既有行为。
@@ -62,3 +62,4 @@ fold 跟踪完整请求标头快照、步骤边界、表层追加与替换、成
- **每次测量都会克隆当前表层**:一致且不可变的快照使读取成为 O(surface),包括低于阈值的压力检查。
- **提供方用量只能为完全相同的规范 envelope 复用**:提示词、前缀、工具、提供方、模型或调用配置变更都会有意回退到完整启发式估算。
- **遗留溯源采取保守策略**:没有 `sourceEventSeqs` 的 assistant 消息无法区分提供方输出与 listener 改写,因此 fold 不会声称已知空流或精确分片流。
- **TUI 与浏览器 fixture 仍保留并行 fold**:`tokenUsage` 拥有持久会话投影语义;TUI 的组合未挂载通用投影 seam,因此继续维护实时的逐步骤 map,而浏览器 fixture 会为独立 demo 数据镜像该单元。

View File

@@ -20,8 +20,8 @@ export interface TokenUsageProjection {
/**
* Approximate context occupancy for a status display.
*
* The two fields are deliberately NOT one atomic request observation:
* `pressureTokens` is the newest provider-reported prompt size in the log,
* The two fields, when present, are deliberately NOT one atomic request
* observation: `pressureTokens` is the newest provider-reported prompt size,
* `contextWindow` the newest recorded route capacity. Switching models can
* therefore pair a fresh capacity with the previous route's pressure until the
* next request reports usage. This is an intentional trade — the value is a
@@ -33,9 +33,9 @@ export interface ContextPressureProjection {
/**
* Provider-reported prompt size of the most recent request: uncached input
* plus cache reads and writes. Response output is excluded, so this does not
* grow as the current turn streams.
* grow as the current turn streams. Absent until a provider reports usage.
*/
pressureTokens: number
pressureTokens?: number
/** Newest recorded route capacity; absent when no adapter advertised one. */
contextWindow?: number
}

View File

@@ -56,10 +56,10 @@ const projectionSchema = z.object({
cacheWriteTokens: z.number().int().nonnegative(),
}).strict()
// Cast for the optional capacity: under exactOptionalPropertyTypes zod infers
// `number | undefined` where the interface declares an absent-or-number field.
// Cast for the optional values: under exactOptionalPropertyTypes zod infers
// `number | undefined` where the interface declares absent-or-number fields.
const pressureSchema = z.object({
pressureTokens: z.number().int().nonnegative(),
pressureTokens: z.number().int().nonnegative().optional(),
contextWindow: z.number().int().positive().optional(),
}).strict() as unknown as z.ZodType<ContextPressureProjection>
@@ -128,12 +128,14 @@ export const contextPressureProjectionDefinition:
ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
key: 'contextPressure',
schema: pressureSchema,
init: () => ({ pressureTokens: 0 }),
init: () => ({}),
apply: (state, event) => {
if (event.type === 'request/context') {
return event.data.contextWindow === state.contextWindow
? state
: { ...state, contextWindow: event.data.contextWindow }
const contextWindow = event.data.contextWindow
if (contextWindow === state.contextWindow) return state
if (contextWindow !== undefined) return { ...state, contextWindow }
const { contextWindow: _removed, ...withoutContextWindow } = state
return withoutContextWindow
}
const usage = event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
? event.data.chunk.usage
@@ -147,5 +149,5 @@ ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
: { ...state, pressureTokens }
},
view: state => state,
stateVersion: 1,
stateVersion: 2,
}

View File

@@ -227,14 +227,25 @@ const pressure = (ctx: Context, session: Session): ContextPressureProjection =>
return value
}
function recordContext(session: Session, model: string, contextWindow: number): void {
session.append('request/context', { provider: 'mock', model, contextWindow })
function recordContext(session: Session, model: string, contextWindow?: number): void {
session.append('request/context', {
provider: 'mock',
model,
...contextWindow === undefined ? {} : { contextWindow },
})
}
describe('contextPressure session projection', () => {
it('serves zero pressure and no capacity for an empty log', async () => {
it('serves no pressure or capacity for an empty log', async () => {
const { ctx, session } = await harness()
expect(pressure(ctx, session)).toEqual({ pressureTokens: 0 })
expect(pressure(ctx, session)).toEqual({})
})
it('does not synthesize zero pressure before a provider usage sample', async () => {
const { ctx, session } = await harness()
startStep(session, 1, 1)
recordContext(session, 'small', 64_000)
expect(pressure(ctx, session)).toEqual({ contextWindow: 64_000 })
})
it('sums prompt-side buckets and excludes response output', async () => {
@@ -271,6 +282,15 @@ describe('contextPressure session projection', () => {
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100, contextWindow: 256_000 })
})
it('removes an older capacity when the newest route advertises none', async () => {
const { ctx, session } = await harness()
startStep(session, 1, 1)
recordContext(session, 'small', 64_000)
usageChunk(session, { inputTokens: 100, outputTokens: 10 }, 1, 1)
recordContext(session, 'unknown')
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100 })
})
it('pushes no change for unrelated events or a restated capacity', async () => {
// The registry gates its change feed on Object.is, so a unit that rebuilt
// state for an event it does not care about would push phantom updates.
@@ -299,6 +319,7 @@ describe('contextPressure session projection', () => {
const checkpoint = JSON.parse(JSON.stringify(
ctx.sessionProjections.checkpoint(session),
)) as ReturnType<typeof ctx.sessionProjections.checkpoint>
expect(checkpoint.contextPressure?.ver).toBe(2)
await meterFiber.dispose()
expect(ctx.sessionProjections.snapshot(session).values).not.toHaveProperty('contextPressure')