docs: rebalance prose cleanup and add trimming skill

This commit is contained in:
Tianyi Cui
2026-07-13 23:27:00 +08:00
parent fcdc318dda
commit 148046b9c8
392 changed files with 2801 additions and 1754 deletions

View File

@@ -22,9 +22,7 @@ The persisted unit IS the existing `SessionEvent` (event-sourced model — the l
## The write coordinator
The two first-party backends were byte-identical (or same-algorithm) for ALL of their write-path orchestration — the in-memory bookkeeping (per-id state, write-behind buffers, per-id serialization chains, per-session init promises), the `session/event` → buffer → `session/flush` drain, lazy materialization, crash-tail repair on load, the four `session/created` adoption cases (new / HMR-adopt / collision / ownerless-claim), and dispose-time quiescence. Only the STORAGE primitives differed (write bytes vs. INSERT rows).
`PersistenceCoordinator` owns that orchestration once. A first-party backend composes one (`new PersistenceCoordinator(ctx, this)`), implements the small `PersistenceBackend` hook interface, and delegates its four public service methods to the coordinator. This keeps the duplicated, correctness-heavy orchestration in a single place (it used to receive the same fixes twice).
`PersistenceCoordinator` owns per-id state, write-behind buffers and serialization, the `session/event` → `session/flush` drain, lazy materialization, crash-tail repair, session adoption, and quiescent disposal. A first-party backend composes one, implements the small `PersistenceBackend` storage hook interface, and delegates its four public service methods. JSONL and SQLite therefore share lifecycle correctness while retaining different storage primitives; see the [coordinator RFC](../../../docs/rfc/implemented/architecture/2026-06-18-shared-persistence-write-coordinator.md).
The `PersistenceBackend<TornMarker>` hooks (the only seam between the coordinator and storage):

View File

@@ -1,6 +1,7 @@
/**
* Shared buffering, serialization, adoption, repair, and disposal orchestration
* over backend-specific persistence primitives.
* for first-party backends. Third-party backends may implement the public
* persistence seam directly.
* @module @deepseek-ai/dsh-session-persistence/coordinator
*/
@@ -84,7 +85,11 @@ interface SessionState {
meta: SessionHeader
/** The next seq the backend expects to append (the stored log length). */
cursor: number
/** Whether lazy creation has produced a durable artifact. */
/**
* Whether lazy creation has produced a durable artifact. The first append
* atomically materializes the header with events; reclaim logic uses this to
* distinguish an unused id from a persisted collision.
*/
materialized: boolean
/**
* The live Session this state was bound to via `onCreated`, if any. State

View File

@@ -1,6 +1,7 @@
/**
* Durable session-persistence seam. Backends store {@link SessionEvent}s plus
* separate {@link SessionHeader} metadata.
* Durable session-persistence seam (`ctx.sessionPersistence`). Backends store
* {@link SessionEvent}s as the event-sourced log and carry non-replayable
* {@link SessionHeader} metadata separately.
* @module @deepseek-ai/dsh-session-persistence
*/
@@ -83,9 +84,9 @@ export abstract class SessionPersistence extends Service {
/**
* Load a header and balanced contiguous log. A complete interrupted final
* turn is preserved and closed with missing tool errors and boundary events;
* only a torn final record is discarded. Unknown versions and corruption in
* the committed prefix reject.
* turn is preserved and durably closed with missing tool errors plus any open
* step and turn boundaries; only a torn final record is discarded. Unknown
* versions and corruption in the committed prefix reject.
* @param id - the persisted session to reload.
* @returns the header and a log ending on a balanced `turn/end`.
*/

View File

@@ -43,8 +43,9 @@ export function oneTurnLog(): SessionEvent[] {
}
/**
* Append a whole event log to a LIVE session, event by event, forwarding the surface metadata
* each event already carries.
* Append recorded events to a live session while forwarding surface metadata verbatim. The broad
* `SessionEvent` union makes the typed marker optional, but the runtime guard must still reject a
* surface event whose fixture omitted it; this helper never synthesizes a default.
*/
export function appendLog(session: Session, events: readonly SessionEvent[]): void {
for (const e of events) {
@@ -217,7 +218,8 @@ export function runPersistenceContract(name: string, make: () => Promise<Contrac
try {
// Every value `isJsonValue` rejects must be rejected by the backend, not just BigInt —
// otherwise a backend could pass this contract while still accepting values that
// corrupt the durable round-trip.
// corrupt the durable round-trip. Each value is carried in a plugin-added field on one
// user message so the contract covers the complete JSON-value boundary.
const cyclic: Record<string, unknown> = { type: 'text', text: 'x' }
cyclic['self'] = cyclic
const badValues: unknown[] = [

View File

@@ -1,5 +1,11 @@
/**
* Reusable ORCHESTRATION suite for any backend that composes a {@link PersistenceCoordinator}.
* Shared write-path orchestration contract for backends using {@link PersistenceCoordinator}.
* Unlike the public storage-semantics suite in `contract.ts`, it covers SessionStore event wiring,
* lazy creation, fork seed persistence, four adoption/collision cases, crash-tail repair, reload,
* flush, and disposal quiescence through public APIs rather than storage primitives.
*
* Each real backend supplies a shared storage scope and optional torn-tail injector; backend specs
* retain only storage-mechanics tests, while these scenarios run once per backend.
* @module @deepseek-ai/dsh-session-persistence/tests/coordinator-contract
*/
@@ -16,10 +22,13 @@ import { meta, oneTurnLog, appendLog } from './contract.ts'
* the suite mounts/disposes backend instances on it and cleans it up at the end.
*/
export interface CoordinatorFixture {
/** Mount a backend over shared fixture storage and return its disposable fiber. */
/** Mount the real backend through `ctx.plugin` over shared storage and return only that fiber. */
mount: (ctx: Context) => Promise<Fiber>
/** Inject an uncommitted torn tail; absent for backends that cannot produce one. */
/**
* Inject a never-committed partial record after the durable region so `loadCore` reaches
* `commitRepair`. Omit only when the backend structurally cannot produce torn tails.
*/
corruptTail?: (id: SessionId, cwd: string | undefined) => Promise<void>
/** Tear down the storage scope (remove the temp dir / file). */
@@ -87,7 +96,7 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise<
it('round-trips the seed boundary (seedLength) through persistence', async () => {
// A forked child records how many leading events were inherited via the seed; the
// boundary must survive a reload (so a resume/replay can tell the inherited prefix from
// the child's own events).
// the child's own events). JSONL stores it in the header; SQLite uses `seed_length`.
const fix = await makeFixture()
const { ctx, fiber } = await freshCtx(fix)
try {
@@ -173,10 +182,10 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise<
})
it('resume: a re-created session seeded with the loaded log does not re-append its seed and continues the seq', async () => {
// Separate backend lifecycles distinguish persisted-seed adoption from an in-memory continuation.
const fix = await makeFixture()
const first = await freshCtx(fix)
try {
// First lifecycle: persist a session through the store.
const s1 = first.ctx.sessions.create(SessionId('resumed'), { meta: { cwd: WORK } })
send(s1, oneTurnLog())
await first.ctx.parallel('session/flush', s1)
@@ -184,9 +193,6 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise<
await first.fiber.dispose()
}
// Second lifecycle: a NEW backend instance + a session re-created with the
// same id SEEDED with the loaded events. onCreated adopts the stored log
// (does not re-persist the seed); a new turn appends at seq 6.
const second = await freshCtx(fix)
try {
const loaded = await second.ctx.sessionPersistence.load(SessionId('resumed'))
@@ -197,7 +203,6 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise<
await second.ctx.parallel('session/flush', s2)
const reloaded = await second.ctx.sessionPersistence.load(SessionId('resumed'))
// 6 original + 2 new, contiguous, no duplicated seed.
expect(reloaded.events.map(e => e.seq)).toEqual([0, 1, 2, 3, 4, 5, 6, 7])
} finally {
await second.fiber.dispose()
@@ -265,7 +270,8 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise<
await ctx.parallel('session/flush', session)
// Hot-reload: dispose instance 1, mount instance 2 over the same storage while the
// session stays live.
// session stays live. The new instance has no coordinator state but must adopt the
// materialized prefix, then persist another turn rather than rejecting it as a collision.
await backend1.dispose()
await fix.mount(ctx)
session.append('turn/start', { turn: 2, trigger: { kind: 'message', source: { kind: 'user' } } })
@@ -358,7 +364,8 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise<
}
// A fresh backend + a NEW live session with the same id but NO explicit resume. onCreated
// treats it as new; create() rejects because a log already exists.
// treats it as new; create() rejects because a log already exists, and `flush()` surfaces
// that initialization rejection.
const second = await freshCtx(fix)
try {
const s2 = second.ctx.sessions.create(SessionId('collide'), { meta: { cwd: WORK } })

View File

@@ -16,8 +16,10 @@ type MemoryStore = Map<string, { meta: SessionHeader; events: SessionEvent[] }>
interface MemoryConfig { store?: MemoryStore }
/**
* A trivial in-memory {@link SessionPersistence} that composes a {@link
* PersistenceCoordinator} over a dependency-free `Map`-backed {@link PersistenceBackend}.
* Reference {@link PersistenceCoordinator} vehicle and abstract-service coverage, backed by a
* dependency-free map with atomic writes and no torn-tail marker. Supplying the map lets multiple
* instances share materialized sessions, the in-memory analogue of reload over one file/database;
* durable behavior is covered by the JSONL and SQLite backends.
*/
class MemoryPersistence extends SessionPersistence implements PersistenceBackend<never> {
static inject = ['sessions']
@@ -78,8 +80,7 @@ class MemoryPersistence extends SessionPersistence implements PersistenceBackend
}
const existing = this.store.get(m.id)
if (!existing) {
// First batch: `_isMaterialized` is false (the coordinator only omits
// materialization on the first batch); writing the entry IS the materialization.
// The coordinator sends the first batch for materialization; later batches append.
this.store.set(m.id, { meta: structuredClone(m), events: structuredClone(events) as SessionEvent[] })
} else {
existing.events.push(...structuredClone(events) as SessionEvent[])
@@ -112,7 +113,8 @@ runPersistenceContract('memory', async () => {
}
})
// Run the shared coordinator orchestration suite against the in-memory backend.
// Each fixture shares one map across mounts. No `corruptTail` is supplied because map writes are
// atomic; the suite asserts that skip while JSONL and SQLite cover the repair branch.
runCoordinatorContract('memory', async (): Promise<CoordinatorFixture> => {
const store: MemoryStore = new Map()
return {