feat(host-runtime,llm-replay): keyless llm seam + replay pacing/consumption handle
BootHostOptions.llm: 'deepseek' | false — false mounts no adapter, boots keyless, and leaves the llm capability seam open for the embedder to fill on RunningHost.ctx (now the third sanctioned ctx use, JSDoc + README amended); an unfilled seam fails loud with NO_ADAPTER at the first stream. dsh-llm-replay grows two additive surfaces for the web browser e2e lane: paceMs (validated per-chunk delay so a real transport shows incremental delivery; abort during a pace wait cancels promptly) and a ReplayHandle return — dispose() plus assertConsumed(), the teardown check that every recorded script bound and drained, converting silent fixture underruns into diagnostics. Existing callers updated; config catalog regenerated.
This commit is contained in:
@@ -24,6 +24,7 @@ Replay keys every call by its calling session id (`GenerateOptions.sessionId`, s
|
||||
| `overrideFile` | string | `$DSH_SNAPSHOT_OVERRIDE` | Optional path to a `ReplayEntry[]` sidecar that replaces the PRIMARY session's derived script. |
|
||||
| `childFiles` | string[] | `$DSH_SNAPSHOT_CHILD_FILES` (path-delimited) | Recorded subagent child-session logs for a nested scenario; empty for a single-session scenario. |
|
||||
| `providers` | `ReplayProviderConfig[]` | — | Optional replay-only provider and model catalog. Each model may publish `contextWindow`; configured routes dispatch through the replay adapter and never perform provider I/O. |
|
||||
| `paceMs` | number | — (burst) | Optional per-chunk delay in ms so downstream transports (e.g. the web SSE mux observed by a real browser) see genuinely incremental delivery. A realism knob only — tests must not depend on it for correctness. Non-negative integer; abort during a pace wait cancels the stream promptly. |
|
||||
|
||||
```yaml
|
||||
- id: llm-replay
|
||||
@@ -43,11 +44,11 @@ Replay keys every call by its calling session id (`GenerateOptions.sessionId`, s
|
||||
|
||||
## Exports
|
||||
|
||||
- `installLlmReplay(ctx, config)` — install the configured replay adapter or catch-all `llm/stream` listener; returns the disposer (HMR safety). Use this in tests to drive replay without the Loader or env vars.
|
||||
- `installLlmReplay(ctx, config)` — install the configured replay adapter or catch-all `llm/stream` listener; returns a `ReplayHandle` (`dispose()` for HMR safety plus `assertConsumed()`, the teardown check that every recorded script bound to a live session and every bound cursor drained — turning a scenario that silently drove fewer model calls than recorded into a crisp diagnostic). Use this in tests to drive replay without the Loader or env vars.
|
||||
- `loadSessionScripts(config)` — resolve the ordered `SessionScript[]` (primary + children) for a scenario, ready to bind to live sessions in first-call order.
|
||||
- `loadReplayScript(config)` — resolve the `ReplayEntry[]` for the PRIMARY session only (sidecar override if present, else derived from the JSONL; fail-loud if the fixture is missing).
|
||||
- `deriveReplayScript(events)` / `parseSessionLog(text)` / `parseSessionHeader(text)` — the pure helpers that turn a recorded session log into a script and read its header `id`/`createdAt`. A derived group must end in a `finish` chunk; a group without one is the fingerprint of a thrown `stream()` and must instead be expressed via an override sidecar.
|
||||
- Types `ReplayEntry` / `SessionScript` / `ReplayConfig` / `ReplayProviderConfig` / `ReplayModelConfig` / `Config`.
|
||||
- Types `ReplayEntry` / `SessionScript` / `ReplayConfig` / `ReplayProviderConfig` / `ReplayModelConfig` / `ReplayHandle` / `Config`.
|
||||
|
||||
## Plugin export shape
|
||||
|
||||
|
||||
@@ -74,6 +74,32 @@ export interface ReplayConfig {
|
||||
* by tests that do not need discovery.
|
||||
*/
|
||||
providers?: ReplayProviderConfig[]
|
||||
/**
|
||||
* Optional per-chunk pacing delay in milliseconds: each replayed chunk waits
|
||||
* this long before yielding, so a downstream transport (e.g. the web SSE
|
||||
* mux observed by a browser) sees genuinely incremental delivery. A realism
|
||||
* knob only — correctness must never depend on it. Absent or `0` keeps
|
||||
* today's synchronous burst yield. Must be a non-negative finite integer;
|
||||
* aborting mid-wait cancels the stream like any other abort.
|
||||
*/
|
||||
paceMs?: number
|
||||
}
|
||||
|
||||
/**
|
||||
* Handle returned by {@link installLlmReplay}: removal plus the end-of-run
|
||||
* consumption check that turns silent fixture underruns (a scenario that
|
||||
* issued fewer calls than recorded, or never bound a recorded child script)
|
||||
* into a crisp diagnostic at teardown.
|
||||
*/
|
||||
export interface ReplayHandle {
|
||||
/** Remove the registered adapter or waterfall listener (HMR safety). Freestanding closure — safe to destructure. */
|
||||
dispose(this: void): void
|
||||
/**
|
||||
* Throw unless every recorded script was bound to a live session and every
|
||||
* bound cursor consumed its full entry list. Call at scenario teardown.
|
||||
* Freestanding closure — safe to destructure.
|
||||
*/
|
||||
assertConsumed(this: void): void
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -277,12 +303,32 @@ class ReplayAdapter extends LlmAdapter {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Wait `paceMs` between chunk yields, aborting the wait (and the stream) the
|
||||
* moment the signal fires — a paced replay must cancel as promptly as a burst
|
||||
* one.
|
||||
*/
|
||||
function paceDelay(paceMs: number, signal: AbortSignal | undefined): Promise<void> {
|
||||
return new Promise<void>((resolve, reject) => {
|
||||
const timer = setTimeout(() => {
|
||||
signal?.removeEventListener('abort', onAbort)
|
||||
resolve()
|
||||
}, paceMs)
|
||||
const onAbort = (): void => {
|
||||
clearTimeout(timer)
|
||||
reject(new Error('aborted'))
|
||||
}
|
||||
signal?.addEventListener('abort', onAbort, { once: true })
|
||||
})
|
||||
}
|
||||
|
||||
/** Yield a recorded stream back, honoring abort like a real adapter. */
|
||||
async function* replayEntry(entry: ReplayEntry, signal: AbortSignal | undefined): AsyncIterable<StreamChunk> {
|
||||
async function* replayEntry(entry: ReplayEntry, signal: AbortSignal | undefined, paceMs: number): AsyncIterable<StreamChunk> {
|
||||
switch (entry.kind) {
|
||||
case 'chunks':
|
||||
for (const chunk of entry.chunks) {
|
||||
if (signal?.aborted) throw new Error('aborted')
|
||||
if (paceMs > 0) await paceDelay(paceMs, signal)
|
||||
yield chunk
|
||||
}
|
||||
return
|
||||
@@ -293,6 +339,7 @@ async function* replayEntry(entry: ReplayEntry, signal: AbortSignal | undefined)
|
||||
// mid-stream STREAM_CLOSED after partial chunks).
|
||||
for (const chunk of entry.chunks) {
|
||||
if (signal?.aborted) throw new Error('aborted')
|
||||
if (paceMs > 0) await paceDelay(paceMs, signal)
|
||||
yield chunk
|
||||
}
|
||||
throw new LlmError(entry.message, entry.code)
|
||||
@@ -319,14 +366,17 @@ async function* replayEntry(entry: ReplayEntry, signal: AbortSignal | undefined)
|
||||
* next ordered recorded script, then advances its own cursor synchronously at
|
||||
* invocation time; calls without `sessionId` share one anonymous session. A
|
||||
* non-empty provider catalog registers a routed replay adapter; otherwise a
|
||||
* catch-all waterfall intercepts requests. Returns the effect disposer for
|
||||
* HMR-safe removal.
|
||||
* catch-all waterfall intercepts requests.
|
||||
*
|
||||
* @param ctx - the context whose LLM service receives the replay route or waterfall.
|
||||
* @param config - the resolved fixture paths (env-var defaulting is `apply`'s job).
|
||||
* @returns the disposer that removes the registered adapter or listener.
|
||||
* @returns the {@link ReplayHandle} carrying the disposer and the teardown consumption check.
|
||||
*/
|
||||
export function installLlmReplay(ctx: Context, config: ReplayConfig): () => void {
|
||||
export function installLlmReplay(ctx: Context, config: ReplayConfig): ReplayHandle {
|
||||
const paceMs = config.paceMs ?? 0
|
||||
if (!Number.isInteger(paceMs) || paceMs < 0) {
|
||||
throw new Error(`llm-replay: paceMs must be a non-negative integer, got ${String(config.paceMs)}`)
|
||||
}
|
||||
const scripts = loadSessionScripts(config)
|
||||
// Live-session → its bound script + cursor. A new live session id claims the
|
||||
// next not-yet-bound script (scripts are in bind order); `nextScript` is the
|
||||
@@ -370,14 +420,31 @@ export function installLlmReplay(ctx: Context, config: ReplayConfig): () => void
|
||||
+ `but its script has only ${boundState.entries.length}; re-record the scenario`,
|
||||
)
|
||||
}
|
||||
yield* replayEntry(entry, options.signal)
|
||||
yield* replayEntry(entry, options.signal, paceMs)
|
||||
})()
|
||||
}
|
||||
const providers = config.providers ?? []
|
||||
if (providers.length > 0) {
|
||||
return ctx.llm.registerAdapter(providers.map(provider => provider.id), new ReplayAdapter(providers, replay))
|
||||
const dispose = providers.length > 0
|
||||
? ctx.llm.registerAdapter(providers.map(provider => provider.id), new ReplayAdapter(providers, replay))
|
||||
: ctx.on('llm/stream', (options: GenerateOptions, _next) => replay(options))
|
||||
return {
|
||||
dispose,
|
||||
assertConsumed(): void {
|
||||
const problems: string[] = []
|
||||
if (nextScript < scripts.length) {
|
||||
problems.push(`${scripts.length - nextScript} recorded script(s) never bound to a live session`)
|
||||
}
|
||||
for (const [key, state] of bound) {
|
||||
if (state.cursor < state.entries.length) {
|
||||
const who = key === ANON ? 'the anonymous session' : `session ${key}`
|
||||
problems.push(`${who} consumed ${state.cursor}/${state.entries.length} recorded call(s)`)
|
||||
}
|
||||
}
|
||||
if (problems.length > 0) {
|
||||
throw new Error(`llm-replay: fixture not fully consumed — ${problems.join('; ')}; the scenario drove fewer model calls than recorded`)
|
||||
}
|
||||
},
|
||||
}
|
||||
return ctx.on('llm/stream', (options: GenerateOptions, _next) => replay(options))
|
||||
}
|
||||
|
||||
export const name = 'llm-replay'
|
||||
@@ -397,6 +464,8 @@ export interface Config {
|
||||
childFiles?: string[]
|
||||
/** Optional replay-only provider catalog; absent or empty selects catch-all waterfall replay. */
|
||||
providers?: ReplayProviderConfig[]
|
||||
/** Optional per-chunk pacing delay in ms (see {@link ReplayConfig.paceMs}); absent keeps burst yield. */
|
||||
paceMs?: number
|
||||
}
|
||||
|
||||
export function apply(ctx: Context, config: Config = {}): void {
|
||||
@@ -413,5 +482,6 @@ export function apply(ctx: Context, config: Config = {}): void {
|
||||
...overrideFile !== undefined && overrideFile.length > 0 ? { overrideFile } : {},
|
||||
...childFiles.length > 0 ? { childFiles } : {},
|
||||
...config.providers !== undefined ? { providers: config.providers } : {},
|
||||
...config.paceMs !== undefined ? { paceMs: config.paceMs } : {},
|
||||
})
|
||||
}
|
||||
|
||||
@@ -234,7 +234,7 @@ describe('installLlmReplay (through the real LlmService)', () => {
|
||||
writeLog(TEXT_CHUNKS)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const dispose = installLlmReplay(ctx, {
|
||||
const { dispose } = installLlmReplay(ctx, {
|
||||
file,
|
||||
providers: [
|
||||
{
|
||||
@@ -429,6 +429,67 @@ describe('installLlmReplay (through the real LlmService)', () => {
|
||||
await iterator.next()
|
||||
await expect(iterator.next()).rejects.toThrow('aborted')
|
||||
})
|
||||
|
||||
it('rejects a paceMs that is not a non-negative integer', async () => {
|
||||
writeLog(TEXT_CHUNKS)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
expect(() => installLlmReplay(ctx, { file, paceMs: -1 })).toThrow(/paceMs/)
|
||||
expect(() => installLlmReplay(ctx, { file, paceMs: 1.5 })).toThrow(/paceMs/)
|
||||
})
|
||||
|
||||
it('paces chunk yields when paceMs is set (each chunk waits at least the pace)', async () => {
|
||||
writeLog(TEXT_CHUNKS)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
installLlmReplay(ctx, { file, paceMs: 10 })
|
||||
const started = performance.now()
|
||||
const chunks = await drain(ctx.llm.stream({ provider: 'm', model: 'm', messages: [] }))
|
||||
expect(chunks).toEqual(TEXT_CHUNKS)
|
||||
// N chunks × 10ms; allow generous scheduling slack, assert the floor only.
|
||||
expect(performance.now() - started).toBeGreaterThanOrEqual(TEXT_CHUNKS.length * 10 - 5)
|
||||
})
|
||||
|
||||
it('aborting DURING a pace wait cancels the stream promptly', async () => {
|
||||
writeLog(TEXT_CHUNKS)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
installLlmReplay(ctx, { file, paceMs: 60_000 })
|
||||
const controller = new AbortController()
|
||||
const pending = drain(ctx.llm.stream({ provider: 'm', model: 'm', messages: [], signal: controller.signal }))
|
||||
// Let the generator park inside the pace timer, then abort — the reject
|
||||
// must come from the abort listener, not the (distant) timer.
|
||||
await new Promise(r => setImmediate(r))
|
||||
controller.abort()
|
||||
await expect(pending).rejects.toThrow('aborted')
|
||||
})
|
||||
|
||||
it('assertConsumed passes only after every recorded call replayed', async () => {
|
||||
writeLog(TEXT_CHUNKS, TEXT_CHUNKS)
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const handle = installLlmReplay(ctx, { file })
|
||||
await drain(ctx.llm.stream({ provider: 'm', model: 'm', messages: [] }))
|
||||
// One of two recorded calls consumed — the underrun must name the gap.
|
||||
expect(() => { handle.assertConsumed() }).toThrow(/consumed 1\/2 recorded call/)
|
||||
await drain(ctx.llm.stream({ provider: 'm', model: 'm', messages: [] }))
|
||||
expect(() => { handle.assertConsumed() }).not.toThrow()
|
||||
})
|
||||
|
||||
it('assertConsumed reports recorded scripts no live session ever bound', async () => {
|
||||
writeLog(TEXT_CHUNKS)
|
||||
const childFile = join(dir, 'session.1.jsonl')
|
||||
writeFileSync(childFile, sessionJsonl(
|
||||
TEXT_CHUNKS.map((chunk, i) => chunkEvent(i + 1, 1, 1, chunk)),
|
||||
{ id: 'child', createdAt: 10 },
|
||||
), 'utf8')
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(LlmService)
|
||||
const handle = installLlmReplay(ctx, { file, childFiles: [childFile] })
|
||||
await drain(ctx.llm.stream({ provider: 'm', model: 'm', messages: [], sessionId: 'live-parent' as NonNullable<GenerateOptions['sessionId']> }))
|
||||
// The child script never bound: the scenario drove fewer sessions than recorded.
|
||||
expect(() => { handle.assertConsumed() }).toThrow(/1 recorded script\(s\) never bound/)
|
||||
})
|
||||
})
|
||||
|
||||
describe('parseSessionHeader', () => {
|
||||
|
||||
Reference in New Issue
Block a user