fix(token-meter): close projected usage review gaps
This commit is contained in:
@@ -475,6 +475,33 @@ interface FixtureTokenUsageProjection {
|
||||
cacheWriteTokens: number
|
||||
}
|
||||
|
||||
interface FixtureUsageSample {
|
||||
turn: number
|
||||
step: number
|
||||
usage: TokenUsage
|
||||
}
|
||||
|
||||
/** Read one provider usage sample from either durable carrier. */
|
||||
function usageSampleOf(event: SessionEvent): FixtureUsageSample | undefined {
|
||||
const item = event as unknown as {
|
||||
type: string
|
||||
data: {
|
||||
turn?: number
|
||||
step?: number
|
||||
usage?: TokenUsage
|
||||
chunk?: { type?: string; usage?: TokenUsage }
|
||||
}
|
||||
}
|
||||
const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage'
|
||||
? item.data.chunk.usage
|
||||
: item.type === 'assistant/message'
|
||||
? item.data.usage
|
||||
: undefined
|
||||
return usage === undefined || item.data.turn === undefined || item.data.step === undefined
|
||||
? undefined
|
||||
: { turn: item.data.turn, step: item.data.step, usage }
|
||||
}
|
||||
|
||||
/** Fixture parallel of token-meter's last-sample-replacing usage projection. */
|
||||
function tokenUsageOf(log: readonly SessionEvent[]): FixtureTokenUsageProjection {
|
||||
const totals: FixtureTokenUsageProjection = {
|
||||
@@ -489,47 +516,40 @@ function tokenUsageOf(log: readonly SessionEvent[]): FixtureTokenUsageProjection
|
||||
buckets: FixtureTokenUsageProjection
|
||||
} | null = null
|
||||
for (const event of log) {
|
||||
const item = event as unknown as {
|
||||
type: string
|
||||
data: {
|
||||
turn?: number
|
||||
step?: number
|
||||
usage?: TokenUsage
|
||||
chunk?: { type?: string; usage?: TokenUsage }
|
||||
}
|
||||
}
|
||||
const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage'
|
||||
? item.data.chunk.usage
|
||||
: item.type === 'assistant/message'
|
||||
? item.data.usage
|
||||
: undefined
|
||||
if (usage === undefined || item.data.turn === undefined || item.data.step === undefined) continue
|
||||
const sample = usageSampleOf(event)
|
||||
if (sample === undefined) continue
|
||||
const buckets: FixtureTokenUsageProjection = {
|
||||
uncachedInputTokens: usage.inputTokens,
|
||||
outputTokens: usage.outputTokens,
|
||||
cacheReadTokens: usage.cacheReadTokens ?? 0,
|
||||
cacheWriteTokens: usage.cacheWriteTokens ?? 0,
|
||||
uncachedInputTokens: sample.usage.inputTokens,
|
||||
outputTokens: sample.usage.outputTokens,
|
||||
cacheReadTokens: sample.usage.cacheReadTokens ?? 0,
|
||||
cacheWriteTokens: sample.usage.cacheWriteTokens ?? 0,
|
||||
}
|
||||
const previous = last?.turn === item.data.turn && last.step === item.data.step
|
||||
const previous = last?.turn === sample.turn && last.step === sample.step
|
||||
? last.buckets
|
||||
: undefined
|
||||
totals.uncachedInputTokens += buckets.uncachedInputTokens - (previous?.uncachedInputTokens ?? 0)
|
||||
totals.outputTokens += buckets.outputTokens - (previous?.outputTokens ?? 0)
|
||||
totals.cacheReadTokens += buckets.cacheReadTokens - (previous?.cacheReadTokens ?? 0)
|
||||
totals.cacheWriteTokens += buckets.cacheWriteTokens - (previous?.cacheWriteTokens ?? 0)
|
||||
last = { turn: item.data.turn, step: item.data.step, buckets }
|
||||
last = { turn: sample.turn, step: sample.step, buckets }
|
||||
}
|
||||
return totals
|
||||
}
|
||||
|
||||
/** Latest log-only capacity record, or undefined before any request ran. */
|
||||
interface FixtureRequestContext {
|
||||
provider: string
|
||||
model: string
|
||||
contextWindow?: number
|
||||
}
|
||||
|
||||
/** Latest log-only route context, or undefined before any request ran. */
|
||||
function lastRequestContext(
|
||||
log: readonly SessionEvent[],
|
||||
): { provider: string; model: string; contextWindow: number } | undefined {
|
||||
): FixtureRequestContext | undefined {
|
||||
const event = log.findLast(item => (item as { type: string }).type === 'request/context')
|
||||
return event === undefined
|
||||
? undefined
|
||||
: (event as unknown as { data: { provider: string; model: string; contextWindow: number } }).data
|
||||
: (event as unknown as { data: FixtureRequestContext }).data
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -539,26 +559,18 @@ function lastRequestContext(
|
||||
*/
|
||||
function contextPressureOf(
|
||||
log: readonly SessionEvent[],
|
||||
): { pressureTokens: number; contextWindow?: number } {
|
||||
let pressureTokens = 0
|
||||
): { pressureTokens?: number; contextWindow?: number } {
|
||||
let pressureTokens: number | undefined
|
||||
for (const event of log) {
|
||||
const item = event as unknown as {
|
||||
type: string
|
||||
data: { usage?: TokenUsage; chunk?: { type?: string; usage?: TokenUsage } }
|
||||
}
|
||||
const usage = item.type === 'assistant/chunk' && item.data.chunk?.type === 'usage'
|
||||
? item.data.chunk.usage
|
||||
: item.type === 'assistant/message'
|
||||
? item.data.usage
|
||||
: undefined
|
||||
if (usage === undefined) continue
|
||||
pressureTokens = usage.inputTokens
|
||||
+ (usage.cacheReadTokens ?? 0)
|
||||
+ (usage.cacheWriteTokens ?? 0)
|
||||
const sample = usageSampleOf(event)
|
||||
if (sample === undefined) continue
|
||||
pressureTokens = sample.usage.inputTokens
|
||||
+ (sample.usage.cacheReadTokens ?? 0)
|
||||
+ (sample.usage.cacheWriteTokens ?? 0)
|
||||
}
|
||||
const contextWindow = lastRequestContext(log)?.contextWindow
|
||||
return {
|
||||
pressureTokens,
|
||||
...pressureTokens === undefined ? {} : { pressureTokens },
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
}
|
||||
}
|
||||
@@ -588,12 +600,7 @@ function projectionValuesOf(log: readonly SessionEvent[]): Record<string, unknow
|
||||
function projectionFramesOf(id: SessionId, log: readonly SessionEvent[], event: SessionEvent): Extract<MuxFrame, { type: 'session/projection' }>[] {
|
||||
const type = (event as { type: string }).type
|
||||
// One usage sample advances both token-meter units.
|
||||
if (
|
||||
(type === 'assistant/chunk'
|
||||
&& (event as unknown as { data: { chunk?: { type?: string } } }).data.chunk?.type === 'usage')
|
||||
|| (type === 'assistant/message'
|
||||
&& (event as unknown as { data: { usage?: TokenUsage } }).data.usage !== undefined)
|
||||
) {
|
||||
if (usageSampleOf(event) !== undefined) {
|
||||
return [
|
||||
{ type: 'session/projection', sessionId: id, key: 'tokenUsage', value: tokenUsageOf(log), seq: event.seq },
|
||||
{ type: 'session/projection', sessionId: id, key: 'contextPressure', value: contextPressureOf(log), seq: event.seq },
|
||||
|
||||
@@ -90,8 +90,8 @@ describe('createFixtureApi', () => {
|
||||
cacheReadTokens: 0,
|
||||
cacheWriteTokens: 0,
|
||||
},
|
||||
// No request ran, so pressure is zero and no capacity is known yet.
|
||||
contextPressure: { pressureTokens: 0 },
|
||||
// No request ran, so neither pressure nor capacity is known yet.
|
||||
contextPressure: {},
|
||||
} },
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/client/ui-conversation/README.md
|
||||
README.md: 84051dbe29503bcb20e317247db2ba2026db4528
|
||||
README.zh.md: 36a0983f95dafc55a4fc76a5bc82112fb369efce
|
||||
README.md: c5b10490abf42bcb04a675d50b3c487fdbf3f3b8
|
||||
README.zh.md: e8bca83ae56c638e7a8fb852c35164b1941d749f
|
||||
|
||||
@@ -22,7 +22,7 @@ Per-session UI state for selection and the active view lives in the declared cha
|
||||
|
||||
The composer bar declares session-scoped single seats for `'conversation.input.plan'` (right of the local access-mode control) and `'conversation.input.model'` (immediately before the pending indicator and send/stop button), plus list slots for overlay, dock, left, and right input extensions. Feature packages own each control and its state; ui-conversation supplies placement, the `locked` owner prop, and the standard slot shares. While the `plan` projection's effective target is plan mode, InputBar swaps its textarea placeholder to the plan-task wording, localized through the `command.hint` locale namespace this package registers and shared verbatim with the claimed `/plan` command hint (a host-folded value read through the standard-kit `useProjection`; owner-supplied placeholders win). A pending composer takeover remains mounted when another conversation view is active so the blocked agent can still receive its answer; without a pending interaction, the active-session composer belongs to Chat. The resident no-session shell uses `DisabledInputBar` and therefore dispatches no session-scoped control seats.
|
||||
|
||||
The chat stats line takes its token accounting from two generic token-meter projections read through the standard-kit `useProjection`: `tokenUsage` for full-log billing (cache hit is `cacheRead / (uncachedInput + cacheRead)`, excluding cache writes) and `contextPressure` for context occupancy. Visible nodes supply only the turn and step counts plus the LLM and tool wall times, which are window-scoped facts about what is on screen rather than accounting. A deployment without token-meter drops the token groups, and a route whose adapter advertises no capacity drops the occupancy group instead of rendering a placeholder. Occupancy is deliberately an approximation — its numerator and capacity are independent last-wins projection fields, not one atomic request observation ([rationale](../../llm/token-meter/README.md)). The inline stats row remains the sole context UI; the model selector has no circle or accessory.
|
||||
The chat stats line takes its token accounting from two generic token-meter projections read through the standard-kit `useProjection`: `tokenUsage` for full-log billing (billed input is uncached input plus cache reads and writes; cache hit divides cache reads by that total) and `contextPressure` for context occupancy. Visible nodes supply only the turn and step counts plus the LLM and tool wall times, which are window-scoped facts about what is on screen rather than accounting; durable token and context groups remain visible when compaction leaves no assistant node in the loaded window. A deployment without token-meter drops the token groups, and occupancy stays hidden until both provider pressure and route capacity are known. Occupancy is deliberately an approximation — its numerator and capacity are independent last-wins projection fields, not one atomic request observation ([rationale](../../llm/token-meter/README.md)). The inline stats row remains the sole context UI; the model selector has no circle or accessory.
|
||||
|
||||
`src/client/` is organized for the future package split: `contract/` is the sole inter-domain shared face (`slots.ts` slot declarations + composed slot props including the tool-row contract, `views.ts` shared primitives, `tool-call-model.ts`); the `skeleton/`, `chat/`, and `toolviews/` (sample registrants) domain directories import contract files and never each other; `apply.ts` is the only assembly point allowed to import all three domains. The `/client` export surface is the contract only — `apply`/`inject`, the two service classes, and the `contract/` type families; implementation components (skeleton, chat rows) and the store factory stay internal and reach the page exclusively through apply's slot registrations (tests take them via the `./src/*` subpath).
|
||||
|
||||
|
||||
@@ -22,7 +22,7 @@ todo 两个面就是在该形状上的两个注册项,都是普通注册方插
|
||||
|
||||
输入栏为 `'conversation.input.plan'`(位于本地 access 模式控件右侧)和 `'conversation.input.model'`(渲染在 pending 指示器与发送/停止按钮之前)声明会话作用域的单实例 seat,并为 overlay、dock、left 和 right 输入扩展声明列表 slot。各功能包拥有相应控件及其状态;ui-conversation 提供放置位置、`locked` owner prop 和标准 slot share。当 `plan` 投影的有效目标为 plan mode 时,InputBar 将文本框 placeholder 切换为 plan 任务措辞,经本包注册的 `command.hint` locale 命名空间本地化,并与已认领 `/plan` 命令的提示逐字共用同一份文案(经标准套件 `useProjection` 读取的 host 折叠值;owner 提供的 placeholder 优先)。另一个会话视图活跃时,待处理的 composer 接管仍保持挂载,使被阻塞的 agent(智能体)仍能收到回答;没有待处理交互时,活跃会话的 composer 归 Chat 所有。常驻无会话壳使用 `DisabledInputBar`,因此不会分发任何会话作用域的控件 seat。
|
||||
|
||||
聊天统计行的 token 账目来自经标准套件 `useProjection` 读取的两个通用 token-meter 投影:`tokenUsage` 提供完整日志计费用量(缓存命中率为 `cacheRead / (uncachedInput + cacheRead)`,不计入缓存写入),`contextPressure` 提供上下文占用率。可见节点只提供轮次与步骤计数,以及 LLM 和工具的墙钟时间:这些是关于「屏幕上有什么」的窗口作用域事实,而非账目。未组合 token-meter 的部署会整组省略 token 分组;适配器未公布容量的路由会省略占用率分组,而不是渲染占位文案。占用率是刻意为之的近似值:它的分子与容量是两个相互独立的「后者胜」投影字段,并非同一次请求的原子观测([原理](../../llm/token-meter/README.md))。行内统计行仍是唯一的上下文 UI;模型选择器不增加圆环或附属控件。
|
||||
聊天统计行的 token 账目来自经标准套件 `useProjection` 读取的两个通用 token-meter 投影:`tokenUsage` 提供完整日志计费用量(计费输入为未缓存输入、缓存读取与缓存写入之和;缓存命中率以缓存读取除以该总量),`contextPressure` 提供上下文占用率。可见节点只提供轮次与步骤计数,以及 LLM 和工具的墙钟时间:这些是关于「屏幕上有什么」的窗口作用域事实,而非账目;压缩使已加载窗口不再包含 assistant 节点时,持久 token 与上下文分组仍保持可见。未组合 token-meter 的部署会整组省略 token 分组;只有提供方压力与路由容量都已知时才显示占用率。占用率是刻意为之的近似值:它的分子与容量是两个相互独立的「后者胜」投影字段,并非同一次请求的原子观测([原理](../../llm/token-meter/README.md))。行内统计行仍是唯一的上下文 UI;模型选择器不增加圆环或附属控件。
|
||||
|
||||
`src/client/` 按未来的包拆分组织:`contract/` 是唯一的跨领域共享表层(`slots.ts` slot 声明 + 组合后的 slot props,包括工具行契约、`views.ts` 共享原语、`tool-call-model.ts`);`skeleton/`、`chat/` 和 `toolviews/`(示例注册方)领域目录只导入 contract 文件,彼此绝不导入;`apply.ts` 是唯一允许导入全部三个领域的组装点。`/client` 导出表层只包含契约:`apply`/`inject`、两个服务类和 `contract/` 类型家族;实现组件(骨架、聊天行)与 store factory 保持内部状态,只能通过 apply 的 slot 注册到达页面(测试通过 `./src/*` 子路径获取它们)。
|
||||
|
||||
|
||||
@@ -78,23 +78,38 @@ export function formatDuration(ms: number): string {
|
||||
* @returns rounded integer percent, or null when no input was billed.
|
||||
*/
|
||||
export function cacheHitPercent(usage: TokenUsageProjection): number | null {
|
||||
const denominator = usage.uncachedInputTokens + usage.cacheReadTokens
|
||||
const denominator = billedInputTokens(usage)
|
||||
return denominator === 0
|
||||
? null
|
||||
: Math.round(usage.cacheReadTokens / denominator * 100)
|
||||
}
|
||||
|
||||
/** Sum the three disjoint prompt-side billing buckets. */
|
||||
function billedInputTokens(usage: TokenUsageProjection): number {
|
||||
return usage.uncachedInputTokens + usage.cacheReadTokens + usage.cacheWriteTokens
|
||||
}
|
||||
|
||||
interface ContextOccupancy {
|
||||
percent: number
|
||||
contextWindow: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Approximate context occupancy, using the TUI's integer rounding and upper
|
||||
* clamp. The numerator and capacity are independent last-wins projection
|
||||
* fields, so this is a reference figure rather than an exact measurement of one
|
||||
* request (see the token-meter README).
|
||||
* @param pressure - the session's context-pressure projection value.
|
||||
* @returns occupancy percent, or null when no capacity is known.
|
||||
* @returns occupancy and its denominator, or null until both values are known.
|
||||
*/
|
||||
export function contextPercent(pressure: ContextPressureProjection | undefined): number | null {
|
||||
if (pressure?.contextWindow === undefined) return null
|
||||
return Math.min(100, Math.round(pressure.pressureTokens / pressure.contextWindow * 100))
|
||||
export function contextOccupancy(
|
||||
pressure: ContextPressureProjection | undefined,
|
||||
): ContextOccupancy | null {
|
||||
if (pressure?.pressureTokens === undefined || pressure.contextWindow === undefined) return null
|
||||
return {
|
||||
percent: Math.min(100, Math.round(pressure.pressureTokens / pressure.contextWindow * 100)),
|
||||
contextWindow: pressure.contextWindow,
|
||||
}
|
||||
}
|
||||
|
||||
/** Props: the conversation-snapshot selector plus the projection read seat. */
|
||||
@@ -108,29 +123,31 @@ export const StatsLine = memo(function StatsLine({ useSession, useProjection }:
|
||||
const usage = useProjection('tokenUsage')
|
||||
const pressure = useProjection('contextPressure')
|
||||
const stats = useMemo(() => deriveStats(nodes), [nodes])
|
||||
if (stats.steps === 0) return null
|
||||
// Pipe-separated groups (figma stats strip); a group with no data drops out whole.
|
||||
const groups: string[] = [`${stats.turns} turns · ${stats.steps} steps`]
|
||||
const durations: string[] = []
|
||||
if (stats.llmMs > 0) durations.push(`LLM ${formatDuration(stats.llmMs)}`)
|
||||
if (stats.toolMs > 0) durations.push(`Tool call ${formatDuration(stats.toolMs)}`)
|
||||
if (durations.length > 0) groups.push(durations.join(' · '))
|
||||
const context = contextPercent(pressure)
|
||||
// Capacity absent (no token-meter composed, or an adapter that advertises
|
||||
// none) drops the group: an unknown denominator has no percentage to show.
|
||||
if (context !== null && pressure?.contextWindow !== undefined) {
|
||||
groups.push(`Context ${context}% of ${formatTokens(pressure.contextWindow)}`)
|
||||
const groups: string[] = []
|
||||
if (stats.steps > 0) {
|
||||
groups.push(`${stats.turns} turns · ${stats.steps} steps`)
|
||||
const durations: string[] = []
|
||||
if (stats.llmMs > 0) durations.push(`LLM ${formatDuration(stats.llmMs)}`)
|
||||
if (stats.toolMs > 0) durations.push(`Tool call ${formatDuration(stats.toolMs)}`)
|
||||
if (durations.length > 0) groups.push(durations.join(' · '))
|
||||
}
|
||||
const context = contextOccupancy(pressure)
|
||||
if (context !== null) {
|
||||
groups.push(`Context ${context.percent}% of ${formatTokens(context.contextWindow)}`)
|
||||
}
|
||||
// Billing rides the durable projection, so these survive paging and
|
||||
// compaction; a deployment without token-meter drops the groups entirely.
|
||||
if (usage !== undefined) {
|
||||
// compaction. Suppress the empty projection on a brand-new session.
|
||||
if (usage !== undefined
|
||||
&& (stats.steps > 0 || billedInputTokens(usage) > 0 || usage.outputTokens > 0)) {
|
||||
const cacheHit = cacheHitPercent(usage)
|
||||
if (cacheHit !== null) groups.push(`Cache hit ${cacheHit}%`)
|
||||
groups.push(
|
||||
`Input ${formatTokens(usage.uncachedInputTokens + usage.cacheReadTokens)} tok`
|
||||
`Input ${formatTokens(billedInputTokens(usage))} tok`
|
||||
+ ` · Output ${formatTokens(usage.outputTokens)} tok`,
|
||||
)
|
||||
}
|
||||
if (groups.length === 0) return null
|
||||
return (
|
||||
<div className={css.root}>
|
||||
{groups.map((group, i) => (
|
||||
|
||||
@@ -218,8 +218,8 @@ describe('small branch tails', () => {
|
||||
})
|
||||
|
||||
it('StatsLine omits the cache-hit segment when no input accounting exists at all', () => {
|
||||
// Cache hit is null only when uncached input and cache reads are both zero
|
||||
// (pure output accounting) — any input makes it a real 0%.
|
||||
// Cache hit is null only when all three prompt buckets are zero (pure
|
||||
// output accounting) — any billed input makes it a real 0%.
|
||||
const snap = {
|
||||
nodes: [{ kind: 'assistant', seq: 1, turn: 1, step: 1, blocks: [], usage: { outputTokens: 10 } }],
|
||||
}
|
||||
|
||||
@@ -122,17 +122,30 @@ describe('StatsLine', () => {
|
||||
return { useSession: bindSnapshotSelector(source), useProjection: projections(values) }
|
||||
}
|
||||
|
||||
it('renders the grouped stats row and hides with zero steps', () => {
|
||||
it('renders the grouped stats row and hides a brand-new empty session', () => {
|
||||
const { source } = makeSource({ nodes: [assistant(1, 1)] })
|
||||
const view = render(<StatsLine {...props(source)} />)
|
||||
// No timing on the fixture: the duration group drops out whole. Tokens come
|
||||
// from the projection, so paging the window cannot change them.
|
||||
expect(view.container.textContent).toBe('1 turns · 1 steps|Cache hit 90%|Input 100 tok · Output 5 tok')
|
||||
const empty = makeSource()
|
||||
const emptyView = render(<StatsLine {...props(empty.source)} />)
|
||||
const emptyView = render(<StatsLine {...props(empty.source, {
|
||||
tokenUsage: { uncachedInputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0 },
|
||||
contextPressure: {},
|
||||
})} />)
|
||||
expect(emptyView.container.textContent).toBe('')
|
||||
})
|
||||
|
||||
it('keeps durable token and context groups after the visible step window is empty', () => {
|
||||
const { source } = makeSource()
|
||||
const view = render(<StatsLine {...props(source, {
|
||||
tokenUsage: USAGE,
|
||||
contextPressure: { pressureTokens: 32_000, contextWindow: 128_000 },
|
||||
})} />)
|
||||
expect(view.container.textContent)
|
||||
.toBe('Context 25% of 128K|Cache hit 90%|Input 100 tok · Output 5 tok')
|
||||
})
|
||||
|
||||
it('renders context occupancy only when the projection knows a capacity', () => {
|
||||
const { source } = makeSource({ nodes: [assistant(1, 1)] })
|
||||
const withCapacity = render(<StatsLine {...props(source, {
|
||||
@@ -146,6 +159,13 @@ describe('StatsLine', () => {
|
||||
contextPressure: { pressureTokens: 32_000 },
|
||||
})} />)
|
||||
expect(noCapacity.container.textContent).not.toContain('Context')
|
||||
// Capacity arrives before usage in the log; no provider sample means there
|
||||
// is no numerator yet, rather than a synthetic 0%.
|
||||
const noPressure = render(<StatsLine {...props(source, {
|
||||
tokenUsage: USAGE,
|
||||
contextPressure: { contextWindow: 128_000 },
|
||||
})} />)
|
||||
expect(noPressure.container.textContent).not.toContain('Context')
|
||||
})
|
||||
|
||||
it('clamps occupancy at 100% when pressure exceeds the recorded capacity', () => {
|
||||
@@ -173,6 +193,20 @@ describe('StatsLine', () => {
|
||||
expect(view.container.textContent).toBe('1 turns · 1 steps|Input 0 tok · Output 7 tok')
|
||||
})
|
||||
|
||||
it('includes cache writes in billed input and the cache-hit denominator', () => {
|
||||
const { source } = makeSource({ nodes: [assistant(1, 1)] })
|
||||
const view = render(<StatsLine {...props(source, {
|
||||
tokenUsage: {
|
||||
uncachedInputTokens: 10,
|
||||
outputTokens: 7,
|
||||
cacheReadTokens: 90,
|
||||
cacheWriteTokens: 100,
|
||||
},
|
||||
})} />)
|
||||
expect(view.container.textContent)
|
||||
.toBe('1 turns · 1 steps|Cache hit 45%|Input 200 tok · Output 7 tok')
|
||||
})
|
||||
|
||||
it('renders ZERO times during streaming chunk frames (RFC hard acceptance)', () => {
|
||||
const { set, source } = makeSource({ nodes: [assistant(1, 1)] })
|
||||
let renders = 0
|
||||
|
||||
@@ -2105,7 +2105,7 @@ export const TYPE_API: readonly TypeApiEntry[] = [
|
||||
},
|
||||
{
|
||||
name: 'RequestContext',
|
||||
declaration: 'export interface RequestContext {\n provider: string;\n model: string;\n contextWindow: number;\n}',
|
||||
declaration: 'export interface RequestContext {\n provider: string;\n model: string;\n contextWindow?: number;\n}',
|
||||
},
|
||||
{
|
||||
name: 'RequestHeaderReason',
|
||||
|
||||
@@ -44,7 +44,7 @@ import {
|
||||
} from '@deepseek-ai/dsh-llm'
|
||||
import type { GenerateOptions, LlmCallConfig, LlmFailure, Message, PreparedLlmCall, ResolvedRetryPolicy } from '@deepseek-ai/dsh-llm'
|
||||
import { canonicalHeader, headerEquals } from '@deepseek-ai/dsh-session'
|
||||
import type { AssistantMessage, Session, SessionId, TurnEndReason, TurnTrigger, UserMessage } from '@deepseek-ai/dsh-session'
|
||||
import type { AssistantMessage, RequestContext, Session, SessionId, TurnEndReason, TurnTrigger, UserMessage } from '@deepseek-ai/dsh-session'
|
||||
import { renderPrompt } from '@deepseek-ai/dsh-system-prompt'
|
||||
import type {} from '@deepseek-ai/dsh-tools'
|
||||
import { executeToolCalls } from './tool-calls.ts'
|
||||
@@ -668,21 +668,21 @@ export class ReactLoopAgent implements Agent {
|
||||
session.append('request/header', { header, reason: 'change' })
|
||||
}
|
||||
|
||||
// Capacity of the route this request resolved to, recorded from the same
|
||||
// Context metadata for the route this request resolved to, recorded from the same
|
||||
// registration-bound lookup that prepared the call (no second resolve).
|
||||
// Deduplicated against the last record: an unchanged route logs nothing.
|
||||
// A route with unknown capacity is still recorded so it clears any older
|
||||
// denominator; an unchanged route logs nothing.
|
||||
const contextWindow = preparedCall?.context?.contextWindow
|
||||
if (contextWindow !== undefined) {
|
||||
const previous = session.requestContext()
|
||||
if (previous?.provider !== config.provider
|
||||
|| previous.model !== config.model
|
||||
|| previous.contextWindow !== contextWindow) {
|
||||
session.append('request/context', {
|
||||
provider: config.provider,
|
||||
model: config.model,
|
||||
contextWindow,
|
||||
})
|
||||
}
|
||||
const requestContext: RequestContext = {
|
||||
provider: config.provider,
|
||||
model: config.model,
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
}
|
||||
const previous = session.requestContext()
|
||||
if (previous?.provider !== requestContext.provider
|
||||
|| previous.model !== requestContext.model
|
||||
|| previous.contextWindow !== requestContext.contextWindow) {
|
||||
session.append('request/context', requestContext)
|
||||
}
|
||||
|
||||
const request = markAgentLoopRequest(deepFreeze({
|
||||
|
||||
@@ -578,13 +578,38 @@ describe('request/context capacity records', () => {
|
||||
.map(event => event.data.contextWindow)).toEqual([64_000, 256_000])
|
||||
})
|
||||
|
||||
it('records nothing when the adapter advertises no capacity', async () => {
|
||||
// The absent-capacity path must stay silent rather than log a placeholder:
|
||||
// consumers read "no capacity known" and omit their percentage entirely.
|
||||
const ctx = await harness(new MockAdapter([textResponse('a')]))
|
||||
it('records and deduplicates a route whose adapter advertises no capacity', async () => {
|
||||
const ctx = await harness(new MockAdapter([textResponse('a'), textResponse('b')]))
|
||||
const agent = ctx.agentLoop.create(SessionId('capacity-absent'), { provider: 'mock', model: 'mock' })
|
||||
send(agent, 'go')
|
||||
send(agent, 'first')
|
||||
await waitForIdle(ctx, agent)
|
||||
expect(agent.session.events.some(event => event.type === 'request/context')).toBe(false)
|
||||
send(agent, 'second')
|
||||
await waitForIdle(ctx, agent)
|
||||
expect(agent.session.events
|
||||
.filter(event => event.type === 'request/context')
|
||||
.map(event => event.data)).toEqual([{ provider: 'mock', model: 'mock' }])
|
||||
})
|
||||
|
||||
it('clears a previous capacity when the next route advertises none', async () => {
|
||||
const adapter = capacityAdapter({ known: 64_000 }, [textResponse('a'), textResponse('b')])
|
||||
const ctx = await harness(adapter)
|
||||
const agent = ctx.agentLoop.create(SessionId('capacity-clear'), { provider: 'mock', model: 'known' })
|
||||
let model = 'known'
|
||||
ctx.on('agent/request', (subject, _turn, _step, _signal, next) => subject === agent
|
||||
? Promise.resolve({ provider: 'mock', model })
|
||||
: next())
|
||||
|
||||
send(agent, 'first')
|
||||
await waitForIdle(ctx, agent)
|
||||
model = 'unknown'
|
||||
send(agent, 'second')
|
||||
await waitForIdle(ctx, agent)
|
||||
|
||||
expect(agent.session.events
|
||||
.filter(event => event.type === 'request/context')
|
||||
.map(event => event.data)).toEqual([
|
||||
{ provider: 'mock', model: 'known', contextWindow: 64_000 },
|
||||
{ provider: 'mock', model: 'unknown' },
|
||||
])
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/core/session/README.md
|
||||
README.md: 861b96c453e677807bded2475fe8e62a74bcd299
|
||||
README.zh.md: 6974a072f5cb32f4e850846bbb02af59cda93303
|
||||
README.md: fe96b5c9735d48d4f92210970b7707749a920787
|
||||
README.zh.md: 6f8aaeef2464a946e10aeef0cfe13b36ab303aeb
|
||||
|
||||
@@ -65,7 +65,7 @@ Providers stream token-sized deltas, so a raw log stores hundreds of `assistant/
|
||||
|
||||
`request/header` records a full canonical snapshot of the non-history request envelope with reason `initial`, `resume`, or `change`. `foldRequestHeader()` selects the latest snapshot; legacy delta events and the removed `fallback` reason are rejected. See the [reconstructable-requests Agent Note](../../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md).
|
||||
|
||||
`request/context` records the registration-bound `contextWindow` of the route a request resolved to, appended inside its step beside `request/header` and only when the provider, model, or capacity differs from the previous record. `session.requestContext()` folds the latest one incrementally, mirroring `requestHeader()`. Capacity stays OUT of `EpochHeader` on purpose: it is adapter metadata describing a route, not an input the request was built from, so it must not enter request reconstruction or header equality — a capacity change is not a header `change`. A route whose adapter advertises no capacity appends nothing.
|
||||
`request/context` records registration-bound metadata for the route a request resolved to, appended inside its step beside `request/header` and only when the provider, model, or capacity differs from the previous record. `session.requestContext()` folds the latest one incrementally, mirroring `requestHeader()`. Capacity stays OUT of `EpochHeader` on purpose: it is adapter metadata describing a route, not an input the request was built from, so it must not enter request reconstruction or header equality — a capacity change is not a header `change`. A route whose adapter advertises no capacity is still recorded with `contextWindow` absent, clearing any older known capacity.
|
||||
|
||||
A `user/message` stores the complete `UserMessage` directly, including the identity created before routing or prompt admission. It renders its `content` verbatim whether it is a direct human prompt, a synthetic injection, or an admitted goal round; its typed `source` is the only channel that tells them apart and carries any domain-specific durable facts. `assistant/message`, `tool/result`, and `steering/message` likewise store complete message values. Turn execution remains enclosed by `turn/start` and `turn/end`, while an idle injection may append and flush a `user/message` between turns without running the model.
|
||||
|
||||
|
||||
@@ -65,7 +65,7 @@
|
||||
|
||||
`request/header` 记录非历史请求封装的完整规范快照,其原因为 `initial`、`resume` 或 `change`。`foldRequestHeader()` 选择最新快照;旧版增量事件和已移除的 `fallback` 原因会被拒绝。详见[可重建请求 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md)。
|
||||
|
||||
`request/context` 记录请求所解析到的路由的、绑定注册项的 `contextWindow`,在其所属步骤内紧随 `request/header` 追加,且仅在提供方、模型或容量与上一条记录不同时追加。`session.requestContext()` 以增量方式归并最新一条,与 `requestHeader()` 保持一致。容量刻意不进入 `EpochHeader`:它是描述路由的适配器元数据,不是构建该请求所依据的输入,因此绝不可进入请求重建或请求头相等性判断:容量变化不构成请求头 `change`。适配器不公布容量的路由不追加任何记录。
|
||||
`request/context` 记录请求所解析到的路由的、绑定注册项的元数据,在其所属步骤内紧随 `request/header` 追加,且仅在提供方、模型或容量与上一条记录不同时追加。`session.requestContext()` 以增量方式归并最新一条,与 `requestHeader()` 保持一致。容量刻意不进入 `EpochHeader`:它是描述路由的适配器元数据,不是构建该请求所依据的输入,因此绝不可进入请求重建或请求头相等性判断:容量变化不构成请求头 `change`。适配器不公布容量的路由仍会被记录,但 `contextWindow` 字段缺失,从而清除较早的已知容量。
|
||||
|
||||
`user/message` 会直接存储完整的 `UserMessage`,其中包括路由或提示词准入前创建的标识。无论它是直接人类提示词、合成注入,还是已准入的 Goal Round,都会原样呈现其 `content`;带类型的 `source` 是区分三者的唯一通道,并携带各领域专有的持久事实。`assistant/message`、`tool/result` 和 steering(中途引导)对应的 `steering/message` 也会存储完整的消息值。轮次执行仍由 `turn/start` 与 `turn/end` 包围,而空闲注入可以在轮次之间追加并刷新一条 `user/message`,无需运行模型。
|
||||
|
||||
|
||||
@@ -584,11 +584,11 @@ export class Session {
|
||||
private contextFoldSeq = 0
|
||||
|
||||
/**
|
||||
* The route capacity in force after the log's last `request/context` event —
|
||||
* The route metadata in force after the log's last `request/context` event —
|
||||
* what the NEXT request deduplicates against — or undefined before any such
|
||||
* record. Maintained incrementally like {@link requestHeader}, so a per-step
|
||||
* read costs O(new events).
|
||||
* @returns the folded capacity record, or undefined when none exists yet.
|
||||
* @returns the folded context record, or undefined when none exists yet.
|
||||
*/
|
||||
requestContext(): RequestContext | undefined {
|
||||
if (this.contextFoldSeq < this.log.length) {
|
||||
|
||||
@@ -170,17 +170,17 @@ export interface EpochHeader {
|
||||
}
|
||||
|
||||
/**
|
||||
* Registration-bound context capacity of one resolved model route. Adapter
|
||||
* Registration-bound context metadata of one resolved model route. Adapter
|
||||
* metadata about a route rather than a request input, which is why it lives
|
||||
* outside {@link EpochHeader}.
|
||||
*/
|
||||
export interface RequestContext {
|
||||
/** Registered provider route the capacity was resolved through. */
|
||||
/** Registered provider route the metadata was resolved through. */
|
||||
provider: string
|
||||
/** Provider-owned model id the capacity belongs to. */
|
||||
/** Provider-owned model id the metadata belongs to. */
|
||||
model: string
|
||||
/** Maximum combined request and response context in tokens. */
|
||||
contextWindow: number
|
||||
/** Maximum combined request and response context in tokens; absent when the adapter advertises none. */
|
||||
contextWindow?: number
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -265,13 +265,13 @@ export interface SessionEventMap {
|
||||
*/
|
||||
'request/header': { header: EpochHeader; reason: RequestHeaderReason }
|
||||
/**
|
||||
* Registration-bound context capacity for the route a request resolved to,
|
||||
* Registration-bound context metadata for the route a request resolved to,
|
||||
* appended inside its step beside `request/header` and only when the route
|
||||
* or capacity differs from the last record. It is log-only and deliberately
|
||||
* NOT part of {@link EpochHeader}: capacity is adapter metadata about a
|
||||
* route, not an input the request was built from, so it must not participate
|
||||
* in request reconstruction or header equality. Absent for a route whose
|
||||
* adapter advertises no capacity.
|
||||
* in request reconstruction or header equality. `contextWindow` is absent
|
||||
* when the route's adapter advertises no capacity.
|
||||
*/
|
||||
'request/context': RequestContext
|
||||
/**
|
||||
|
||||
@@ -96,7 +96,7 @@ describe('Session.requestContext', () => {
|
||||
const CAPACITY = { provider: 'mock', model: 'm', contextWindow: 128_000 }
|
||||
|
||||
/** A turn-enclosed capacity record; the invariant rejects one outside a turn. */
|
||||
function seedWith(...records: { provider: string; model: string; contextWindow: number }[]): SessionEvent[] {
|
||||
function seedWith(...records: { provider: string; model: string; contextWindow?: number }[]): SessionEvent[] {
|
||||
const events: SessionEvent[] = [{
|
||||
type: 'turn/start', seq: 0, time: 1, data: { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } },
|
||||
}]
|
||||
@@ -127,6 +127,8 @@ describe('Session.requestContext', () => {
|
||||
expect(session.requestContext()).toEqual(CAPACITY)
|
||||
session.append('request/context', { ...CAPACITY, model: 'next', contextWindow: 64_000 })
|
||||
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'next', contextWindow: 64_000 })
|
||||
session.append('request/context', { provider: 'mock', model: 'unknown' })
|
||||
expect(session.requestContext()).toEqual({ provider: 'mock', model: 'unknown' })
|
||||
})
|
||||
|
||||
it('folds a batch appended between two reads', () => {
|
||||
@@ -143,6 +145,6 @@ describe('Session.requestContext', () => {
|
||||
const held = session.requestContext()
|
||||
if (held === undefined) throw new Error('expected a folded capacity record')
|
||||
expect(Object.isFrozen(held)).toBe(true)
|
||||
expect(() => { (held as { contextWindow: number }).contextWindow = 1 }).toThrow()
|
||||
expect(() => { (held as { contextWindow?: number }).contextWindow = 1 }).toThrow()
|
||||
})
|
||||
})
|
||||
|
||||
@@ -2,5 +2,5 @@
|
||||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/llm/token-meter/README.md
|
||||
README.md: b9bf1dfa253e424e5ec35cd3e7bf0f52af579077
|
||||
README.zh.md: c97fc0b87364dfe9ca46139f0ec82519e191b772
|
||||
README.md: 701893b342f9a93a75bec175634b1054f3d17151
|
||||
README.zh.md: a5844e8788422bba669632ed587fb87e1e2a1e58
|
||||
|
||||
@@ -27,7 +27,7 @@ When the composition provides `ctx.sessionProjections`, token-meter registers tw
|
||||
|
||||
`tokenUsage` carries the complete durable log's `uncachedInputTokens`, `outputTokens`, `cacheReadTokens`, and `cacheWriteTokens`. Usage chunks are counted even when a request later fails; a final assistant-message usage for the same `(turn, step)` replaces that sample instead of double-counting it. Reasoning remains an output subdivision. The single last-sample slot relies on a session-log ordering property: once a later step reports usage, a legal log never reports usage for an earlier step again.
|
||||
|
||||
`contextPressure` carries `pressureTokens` — the newest provider-reported prompt size, summing uncached input plus cache reads and writes — and the optional `contextWindow` from the newest `request/context` record. Output is excluded, so the numerator holds still while a turn streams and steps forward when the next request reports its usage.
|
||||
`contextPressure` carries optional `pressureTokens` — the newest provider-reported prompt size, summing uncached input plus cache reads and writes — and optional `contextWindow` from the newest `request/context` record. Pressure stays absent until a provider reports usage; capacity stays absent for a route whose adapter advertises none. Output is excluded, so the numerator holds still while a turn streams and steps forward when the next request reports its usage.
|
||||
|
||||
Both units use the standard projection baseline, live frame, higher-seq-wins store, and JSON checkpoint paths. Unloading token-meter removes both keys. A headless or TUI composition without the projection seam keeps the measurement service's existing behavior.
|
||||
|
||||
@@ -62,3 +62,4 @@ No direct invalidation; the named consumer owns any request-prefix changes.
|
||||
- **Every measurement clones the current surface** — coherent immutable snapshots make reads O(surface), including below-threshold pressure checks.
|
||||
- **Provider usage is only reusable for an identical canonical envelope** — prompt, prefix, tools, provider, model, or call-config changes deliberately fall back to full heuristic estimation.
|
||||
- **Legacy provenance is conservative** — assistant messages without `sourceEventSeqs` cannot distinguish provider output from listener rewrites, so the fold avoids claiming a known empty or exact chunk stream.
|
||||
- **The TUI and browser fixture retain parallel folds** — `tokenUsage` owns durable session-projection semantics; the TUI keeps its live per-step map because its composition does not mount the generic projection seam, while the browser fixture mirrors the unit for standalone demo data.
|
||||
|
||||
@@ -27,7 +27,7 @@ fold 跟踪完整请求标头快照、步骤边界、表层追加与替换、成
|
||||
|
||||
`tokenUsage` 携带完整持久日志中的 `uncachedInputTokens`、`outputTokens`、`cacheReadTokens` 和 `cacheWriteTokens`。即使请求随后失败,用量分片仍会计入;同一 `(turn, step)` 的最终 assistant 消息用量会替换该样本,而不是重复计数。推理仍是输出的一个细分项。只保留单个最新样本,依赖的是会话日志的一条顺序性质:一旦某个更晚的步骤报告了用量,合法日志就绝不会再为更早的步骤报告用量。
|
||||
|
||||
`contextPressure` 携带 `pressureTokens`(提供方报告的最新提示词规模,为未缓存输入加缓存读取与写入之和),以及来自最新一条 `request/context` 记录的可选 `contextWindow`。输出不计入其中,因此轮次流式输出期间分子保持不动,等到下一个请求报告用量时才前进。
|
||||
`contextPressure` 携带可选的 `pressureTokens`(提供方报告的最新提示词规模,为未缓存输入加缓存读取与写入之和),以及来自最新一条 `request/context` 记录的可选 `contextWindow`。提供方报告用量前压力保持缺失;路由适配器未公布容量时容量也保持缺失。输出不计入其中,因此轮次流式输出期间分子保持不动,等到下一个请求报告用量时才前进。
|
||||
|
||||
两个单元都使用标准的投影基线、实时帧、seq 高者胜值仓和 JSON 检查点路径。卸载 token-meter 会移除这两个键。不带投影 seam 的 headless 或 TUI 组合会保留测量服务的既有行为。
|
||||
|
||||
@@ -62,3 +62,4 @@ fold 跟踪完整请求标头快照、步骤边界、表层追加与替换、成
|
||||
- **每次测量都会克隆当前表层**:一致且不可变的快照使读取成为 O(surface),包括低于阈值的压力检查。
|
||||
- **提供方用量只能为完全相同的规范 envelope 复用**:提示词、前缀、工具、提供方、模型或调用配置变更都会有意回退到完整启发式估算。
|
||||
- **遗留溯源采取保守策略**:没有 `sourceEventSeqs` 的 assistant 消息无法区分提供方输出与 listener 改写,因此 fold 不会声称已知空流或精确分片流。
|
||||
- **TUI 与浏览器 fixture 仍保留并行 fold**:`tokenUsage` 拥有持久会话投影语义;TUI 的组合未挂载通用投影 seam,因此继续维护实时的逐步骤 map,而浏览器 fixture 会为独立 demo 数据镜像该单元。
|
||||
|
||||
@@ -20,8 +20,8 @@ export interface TokenUsageProjection {
|
||||
/**
|
||||
* Approximate context occupancy for a status display.
|
||||
*
|
||||
* The two fields are deliberately NOT one atomic request observation:
|
||||
* `pressureTokens` is the newest provider-reported prompt size in the log,
|
||||
* The two fields, when present, are deliberately NOT one atomic request
|
||||
* observation: `pressureTokens` is the newest provider-reported prompt size,
|
||||
* `contextWindow` the newest recorded route capacity. Switching models can
|
||||
* therefore pair a fresh capacity with the previous route's pressure until the
|
||||
* next request reports usage. This is an intentional trade — the value is a
|
||||
@@ -33,9 +33,9 @@ export interface ContextPressureProjection {
|
||||
/**
|
||||
* Provider-reported prompt size of the most recent request: uncached input
|
||||
* plus cache reads and writes. Response output is excluded, so this does not
|
||||
* grow as the current turn streams.
|
||||
* grow as the current turn streams. Absent until a provider reports usage.
|
||||
*/
|
||||
pressureTokens: number
|
||||
pressureTokens?: number
|
||||
/** Newest recorded route capacity; absent when no adapter advertised one. */
|
||||
contextWindow?: number
|
||||
}
|
||||
|
||||
@@ -56,10 +56,10 @@ const projectionSchema = z.object({
|
||||
cacheWriteTokens: z.number().int().nonnegative(),
|
||||
}).strict()
|
||||
|
||||
// Cast for the optional capacity: under exactOptionalPropertyTypes zod infers
|
||||
// `number | undefined` where the interface declares an absent-or-number field.
|
||||
// Cast for the optional values: under exactOptionalPropertyTypes zod infers
|
||||
// `number | undefined` where the interface declares absent-or-number fields.
|
||||
const pressureSchema = z.object({
|
||||
pressureTokens: z.number().int().nonnegative(),
|
||||
pressureTokens: z.number().int().nonnegative().optional(),
|
||||
contextWindow: z.number().int().positive().optional(),
|
||||
}).strict() as unknown as z.ZodType<ContextPressureProjection>
|
||||
|
||||
@@ -128,12 +128,14 @@ export const contextPressureProjectionDefinition:
|
||||
ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
|
||||
key: 'contextPressure',
|
||||
schema: pressureSchema,
|
||||
init: () => ({ pressureTokens: 0 }),
|
||||
init: () => ({}),
|
||||
apply: (state, event) => {
|
||||
if (event.type === 'request/context') {
|
||||
return event.data.contextWindow === state.contextWindow
|
||||
? state
|
||||
: { ...state, contextWindow: event.data.contextWindow }
|
||||
const contextWindow = event.data.contextWindow
|
||||
if (contextWindow === state.contextWindow) return state
|
||||
if (contextWindow !== undefined) return { ...state, contextWindow }
|
||||
const { contextWindow: _removed, ...withoutContextWindow } = state
|
||||
return withoutContextWindow
|
||||
}
|
||||
const usage = event.type === 'assistant/chunk' && event.data.chunk.type === 'usage'
|
||||
? event.data.chunk.usage
|
||||
@@ -147,5 +149,5 @@ ProjectionDefinition<'contextPressure', ContextPressureProjection> = {
|
||||
: { ...state, pressureTokens }
|
||||
},
|
||||
view: state => state,
|
||||
stateVersion: 1,
|
||||
stateVersion: 2,
|
||||
}
|
||||
|
||||
@@ -227,14 +227,25 @@ const pressure = (ctx: Context, session: Session): ContextPressureProjection =>
|
||||
return value
|
||||
}
|
||||
|
||||
function recordContext(session: Session, model: string, contextWindow: number): void {
|
||||
session.append('request/context', { provider: 'mock', model, contextWindow })
|
||||
function recordContext(session: Session, model: string, contextWindow?: number): void {
|
||||
session.append('request/context', {
|
||||
provider: 'mock',
|
||||
model,
|
||||
...contextWindow === undefined ? {} : { contextWindow },
|
||||
})
|
||||
}
|
||||
|
||||
describe('contextPressure session projection', () => {
|
||||
it('serves zero pressure and no capacity for an empty log', async () => {
|
||||
it('serves no pressure or capacity for an empty log', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 0 })
|
||||
expect(pressure(ctx, session)).toEqual({})
|
||||
})
|
||||
|
||||
it('does not synthesize zero pressure before a provider usage sample', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
startStep(session, 1, 1)
|
||||
recordContext(session, 'small', 64_000)
|
||||
expect(pressure(ctx, session)).toEqual({ contextWindow: 64_000 })
|
||||
})
|
||||
|
||||
it('sums prompt-side buckets and excludes response output', async () => {
|
||||
@@ -271,6 +282,15 @@ describe('contextPressure session projection', () => {
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100, contextWindow: 256_000 })
|
||||
})
|
||||
|
||||
it('removes an older capacity when the newest route advertises none', async () => {
|
||||
const { ctx, session } = await harness()
|
||||
startStep(session, 1, 1)
|
||||
recordContext(session, 'small', 64_000)
|
||||
usageChunk(session, { inputTokens: 100, outputTokens: 10 }, 1, 1)
|
||||
recordContext(session, 'unknown')
|
||||
expect(pressure(ctx, session)).toEqual({ pressureTokens: 100 })
|
||||
})
|
||||
|
||||
it('pushes no change for unrelated events or a restated capacity', async () => {
|
||||
// The registry gates its change feed on Object.is, so a unit that rebuilt
|
||||
// state for an event it does not care about would push phantom updates.
|
||||
@@ -299,6 +319,7 @@ describe('contextPressure session projection', () => {
|
||||
const checkpoint = JSON.parse(JSON.stringify(
|
||||
ctx.sessionProjections.checkpoint(session),
|
||||
)) as ReturnType<typeof ctx.sessionProjections.checkpoint>
|
||||
expect(checkpoint.contextPressure?.ver).toBe(2)
|
||||
|
||||
await meterFiber.dispose()
|
||||
expect(ctx.sessionProjections.snapshot(session).values).not.toHaveProperty('contextPressure')
|
||||
|
||||
Reference in New Issue
Block a user