From 0056238bbd19b4c824883a84fda5493f229e6f53 Mon Sep 17 00:00:00 2001 From: Ziya <199893125+ZiyaZhang@users.noreply.github.com> Date: Tue, 7 Jul 2026 01:01:24 -0700 Subject: [PATCH 1/4] =?UTF-8?q?docs(rfc):=20propose=20recallable=20compact?= =?UTF-8?q?ion=20=E2=80=94=20split=20checkpoints=20and=20in-session=20hist?= =?UTF-8?q?ory=20recall?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/rfc/INDEX.md | 1 + .../2026-07-06-recallable-compaction.md | 106 ++++++++++++++++++ 2 files changed, 107 insertions(+) create mode 100644 docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md diff --git a/docs/rfc/INDEX.md b/docs/rfc/INDEX.md index bcd78d8b88..7bfda3401e 100644 --- a/docs/rfc/INDEX.md +++ b/docs/rfc/INDEX.md @@ -12,6 +12,7 @@ Generated by `pnpm run gen-rfc-index` from the RFC tree — never edit by hand; | [Multiplex concurrent ACP sessions over one connection](proposed/feature/2026-06-14-acp-multi-session.md) | 2026-06-14 | | [Optional Code Mode — model writes TypeScript against an SDK of all tools](proposed/feature/2026-06-15-optional-code-mode.md) | 2026-06-15 | | [Pre-tool input rewrite — a consistent design](proposed/feature/2026-06-30-pre-tool-input-rewrite.md) | 2026-06-30 | +| [Recallable compaction — index checkpoints, a state checkpoint, and in-session history recall](proposed/feature/2026-07-06-recallable-compaction.md) | 2026-07-06 | ### Simplification diff --git a/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md b/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md new file mode 100644 index 0000000000..302026e33f --- /dev/null +++ b/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md @@ -0,0 +1,106 @@ +# RFC: Recallable compaction — index checkpoints, a state checkpoint, and in-session history recall + +Status: proposed + +## Problem + +Compaction is a one-way door. The summary the model sees carries no reference to what it shadows — the `shadowedRange` provenance lives only on the log-only `compact/summary` event — and no tool lets the model read a shadowed span back. Whatever the summarizer drops is gone from the model's reachable world, even though the append-only log holds every byte. Repeated compaction compounds this: the head checkpoint is rewritten every pass, so the request prefix takes a full prompt-cache miss each time, and earlier summaries are re-summarized generation after generation. + +The root cause is one artifact playing two conflicting roles. An **index** wants to be frozen, chronological, and cheap; the model's **working memory** wants a global view, re-prioritization, and mutability. A single summary can be neither well. + +No mainstream coding harness gives the model in-loop recall, and none of the surveyed implementations makes compaction prefix-cache-aware. An event-sourced session — originals durable, seq-addressable, replay-exact — is the natural substrate for both. + +## Proposal + +Split the checkpoint into two classes and make shadowed history reachable. + +### Frozen index checkpoints + +Newly stale history splits into chunks by deterministic policy: accumulate toward `chunkTokens`, snap edges to balanced tool-pairing cuts (`isToolPairingBalanced`), prefer turn boundaries, and place the final boundary as close to the retain boundary as balance allows, so the trailing slice shrinks to roughly one turn. Each chunk is compacted by one `compactRegion` call into an **index stub** (`stubTokens`, ~100–200 tokens): + +- two or three lines of what happened; +- a keyword line of low-frequency literal anchors — exact error strings, values, config keys — grouped by kind; +- a code-composed footer: `[checkpoint c: shadows conversation span #–#; originals retrievable via history_read]`. Pointers are assembled from provenance, never model-authored. + +A committed stub is never rewritten and never re-enters a later compaction region. A slice consisting of recalled content is stubbed by code alone — a pointer line, no LLM call. A failed stub call degrades the same way: its slice gets a code-only pointer stub and the pass continues, making the state rewrite the only hard LLM dependency in a pass. + +### The state checkpoint + +One mutable working-memory document (at most one; zero before the first pass), positioned after all stubs and before the retained tail. Each pass rewrites it from the previous state plus this pass's staled content — O(previous + new), under the merge-don't-restate rule already in the summarization prompt — covering decisions, current state, constraints, and next steps. It carries its own footer and a size cap at the scale of today's summary. Stub calls receive the pass-start state as background, context only. + +An inflation guard bounds the whole pass: if the post-compaction size is not strictly below the pre-compaction size, nothing commits and the turn proceeds; the attempt defers until more stale history accumulates. The guard compares one metric on both sides — provider-reported usage (the counters PR #197 introduces), falling back to the character estimator on both sides. + +### Pass execution + +- Chunk slices are surface position ranges. A pass runs two phases: all summarize calls execute concurrently, buffered off-surface; then regions commit strictly left to right — chunks first, trailing slice last — so the state checkpoint lands after every stub through contiguous single-node replaces. Wall-clock stays near one summarize call. +- The superseded state checkpoint folds into the next pass's first chunk as ordinary history: no tombstone, no new primitive. Its stub omits it, `history_read` renders it labeled `[prior state checkpoint]`, and its footer travels with the rendered text, keeping every trailing slice reachable through the two-hop chain. +- Range selection is frozen-aware: the compactable span begins after the last committed index checkpoint, at the surface head only when none exists. A legacy session's existing head checkpoint is adopted as state-class — its text the merge base, its node folded like any superseded state. +- A crash in the summarize phase commits nothing; a crash mid-commit leaves a left-to-right prefix committed, and the resumed pass reads its merge base from the log's latest state-class `compact/summary` event and commits the remaining regions unconditionally — restoring `[stubs…][state][tail]` outranks shrinking. + +### The recall tools + +A new package `@deepseek-ai/dsh-tool-recall` (consumer-only, over the `dsh-session` and `dsh-compact` vocabularies) registers two model-facing tools: + +- `history_read(checkpoint, offset?)` — renders the shadowed span of any checkpoint in the log, including superseded ones, as `User:`/`Assistant:`/`Tool result:` transcript, paginated by a configured budget with a continuation cursor. +- `history_search(query, checkpoint?, limit?)` — case-insensitive literal scan over every shadowed span; returns snippets with checkpoint ids and coverage metadata (`scanned`/`matched`/`truncated`). The zero-match hint notes the scan is literal and points at direct `history_read` of a plausible checkpoint. + +Both read `exec.agent.session.events` (the tool-todo access pattern; non-agent callers rejected), render only surface-type message events, and return ordinary `tool/result`s — recalled bytes land at the context tail, logged, so reconstructability holds with no special casing. There is no new storage and no sidecar index: the session log is the archive, `compact/summary` provenance is the index metadata, and the tools are a read path over both. The tool schemas and the package's one system-prompt section are static strings; checkpoint ids reach the model only through footers. The transcript renderer moves from `compact-basic` into `dsh-session`, shared by summarizer and tools. + +### Cache and cost + +The request prefix after a pass is `[system][stubs…][state][tail]`. Frozen stubs are byte-stable across passes, so the miss begins at the token replacing the previous state checkpoint and stays O(new chunks + state + tail) — against position zero today. Recall output lands at the tail, leaving the prefix untouched. Per-pass summarize input is roughly twice today's plus an m·S background term, bounded by a `chunkTokens` floor (a small multiple of the state cap) and a validated `stubTokens`/`chunkTokens` ratio ceiling; a shared-prefix input layout (preamble, then the byte-identical pass-start state, slice content in the tail) lets sibling calls earn cached-rate rereads. + +### Packaging + +The design ships as a new backend `dsh-compact-recallable` on the existing `ctx.compact` seam, enabled by default in the shipped example configs; `compact-basic` remains as the reference implementation and the seam's design twin, in the pattern of the paired LLM adapters. The seam JSDoc's "at most one auto-generated checkpoint, always at the head" clause is relaxed to name both backend behaviors. + +### Relation to in-flight work + +- **Tool-result pruning (PR #113)**: its replacement nodes carry `sourceEventSeqs`; the same registry fold lists pruned results as recallable. Follow-up scope; neither blocks the other. +- **Provider-usage token pressure (PR #197)**: supplies the guard's accounting; the implementation stacks after it. +- **"Query sessions" backlog item**: the cross-session generalization; this RFC scopes to the live session with tool names and rendering chosen so that work extends rather than collides. +- **Training**: when to recall is a learned behavior. The deterministic footers and keyword anchors give training a stable target, and recall usage is fully visible in the session log for trajectory export; benchmark and RL design proceed with the post-training side. + +### Follow-ups + +Specified during review, deferred until observation calls for them: + +- Guard degradation ladder (code-only rollup of the oldest stub prefix, footers preserved, rolled-up ids remain recall targets; then one summary after the frozen boundary) — on observed guard livelock or stub-region pressure. +- Echo detection on stub outputs (sentence-scale n-grams, short literals exempt, retry then strip) — on observed division-of-labor leakage. +- Periodic state refresh from chunk originals — on observed drift in the handoff probe. +- `stateFallbackThreshold` (full-detail state prompt below a stub count) — on short-session regression. +- Lazy registration of the recall tools — on measured context tax in never-compacting sessions. +- Split summarizer models; model-chosen chunk boundaries; cross-session recall; semantic search fallback — each behind its own evidence. + +## Alternatives considered + +- **Staged delivery** (ship recall tools alone over today's backend; gate the checkpoint split on observed recall usage) — rejected: untrained models under-use any new tool, so the gate would measure training absence rather than design value, while the training side needs the complete mechanism to build environments against; the pre-release window is when persisted-format changes are cheapest; and the cache economics are first-party knowledge, not a hypothesis awaiting telemetry. The implementation still lands as stacked PRs with the recall tools first — construction order, not a decision gate. +- **All-frozen full-size summaries, no state checkpoint** — rejected: unbounded permanent-prefix growth, self-accelerating toward thrashing, with nothing left to re-prioritize. +- **Pure stubs, no state checkpoint** — rejected: presumes the model knows what it is missing; fails on unknown unknowns. +- **LLM aging/consolidation of frozen chunks** — rejected as a routine mechanism: summary-of-summary loss and frozen-prefix churn; the code-only rollup is its surviving form, deferred. +- **Full prefix as chunk-summarizer input** — rejected: O(N²); the state document gives the same background at O(state). +- **One summarize call emitting all outputs** — rejected: the summarize path has no structured-output enforcement; parsing one free-text response apart is the fragile seam the fail-closed design avoids. +- **Model-chosen chunk boundaries** — deferred: parse-and-validate cost against unproven value; chunk policy sits behind config. +- **Model-authored pointers** — rejected: pointers must be exact; deterministic assembly is. +- **FTS/vector index sidecar** — rejected in-session: the live log is in memory and bounded, a literal scan under budget suffices; an index earns its keep at cross-session scope. +- **Semantic search fallback / secondary-model extraction in the recall path** — rejected: an LLM or embedding call there breaks keyless replay determinism; recall stays a pure function of the log. +- **Raw events instead of rendered transcript** — rejected: leaks log-only vocabulary and chunk noise; the model reads what a model once saw. +- **Doing nothing (resume/fork as recovery)** — rejected: it makes recovery a human act. + +## Acceptance criteria + +- Auto-compaction over a long session yields `[stubs…][state][tail]` after every completed pass; prior stubs stay byte-identical across passes; committed stubs never fall inside a later region; the superseded state checkpoint folds without a tombstone, renders labeled, and stays reachable and searchable through the two-hop chain. +- Every checkpoint's surface text ends with the deterministic footer; footers round-trip through replay byte-identically; the state checkpoint's provenance records its wider input range. +- Nothing commits before all summaries exist and the guard passes on like-for-like accounting; a guard failure commits nothing and does not fail the turn; a mid-commit kill resumed at the next pre-step completes the pass with the state region committed unconditionally, merge base read from the log; a legacy head checkpoint is adopted as state-class. +- `history_read` renders any logged checkpoint's span under budget with a working cursor; `history_search` covers every shadowed span with checkpoint-id snippets and coverage metadata, asserted in particular by finding content that exists only in a span shadowed by a superseded state checkpoint — the regression pin for trailing-slice reachability; both reject non-agent callers and never-existing ids or orphaned `compact/start` with typed errors; recalled content appears as ordinary `tool/result`s; request-reconstruction invariants pass over sessions with compaction plus recall; one keyless snapshot scenario covers compact-then-recall end to end; tool schemas and the prompt section are byte-identical across passes. +- On the long-horizon bench suite: task success does not regress against `compact-basic` at equal budgets; a handoff-fidelity probe (restate K known decisions and constraints after a pass) scores no worse; recall usage frequency and hit usefulness are reported per run via the dsh bench report pipeline, alongside the stub-directory attention measurement and cache-hit telemetry. +- Seam JSDoc, the compaction capability-seam RFC, `architecture.md`, and the generated tool, config, persistence, and module-graph catalogs update in the same change; all budgets live in config; new source directories hold per-file 100% coverage with HMR disposal tests. + +## Risks + +- **Recall is a learned behavior**: untrained models will under-use it, and the bench report exists to track the gap while training closes it. Until then the state checkpoint keeps the floor at today's summary quality. +- **Unknown unknowns remain**: a detail absent from summaries and keywords draws no recall. Recall converts "unreachable even when suspected" into "reachable when suspected". +- **The stub directory occupies attention**: dozens of stable index cards per request may dilute focus; the bench measurement in the acceptance criteria tracks it against `compact-basic`. +- **Cost**: per-pass summarize input is roughly twice today's; short sessions sit near today's cost and quality, and the design pays off with session length. +- **State drift and division-of-labor leakage** are observable through the handoff probe and stub review; their counters are specified follow-ups. +- **Two backends** are a maintenance surface; the seam contract and the shared recall consumer bound it, and the bench comparison decides the default over time. From ac64a0c7b123df217618a3581c04f36bcea8ab9e Mon Sep 17 00:00:00 2001 From: Ziya <199893125+ZiyaZhang@users.noreply.github.com> Date: Tue, 7 Jul 2026 02:58:56 -0700 Subject: [PATCH 2/4] docs(rfc): specify stub input layering; add amortized stub drafting follow-up --- .../rfc/proposed/feature/2026-07-06-recallable-compaction.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md b/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md index 302026e33f..e657887fe1 100644 --- a/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md +++ b/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md @@ -22,11 +22,11 @@ Newly stale history splits into chunks by deterministic policy: accumulate towar - a keyword line of low-frequency literal anchors — exact error strings, values, config keys — grouped by kind; - a code-composed footer: `[checkpoint c: shadows conversation span #–#; originals retrievable via history_read]`. Pointers are assembled from provenance, never model-authored. -A committed stub is never rewritten and never re-enters a later compaction region. A slice consisting of recalled content is stubbed by code alone — a pointer line, no LLM call. A failed stub call degrades the same way: its slice gets a code-only pointer stub and the pass continues, making the state rewrite the only hard LLM dependency in a pass. +A committed stub is never rewritten and never re-enters a later compaction region. A stub call's input is layered: the fixed preamble and the byte-identical pass-start state checkpoint (the shared prefix across all calls in the phase), then the keyword lines of all previously committed stubs — so a new entry indexes what is distinctive to its chunk instead of repeating the directory — the one or two most recent committed stubs for chronological continuity, and the slice itself. Sibling stubs from the same pass are not inputs (the concurrent phase forbids it; turn-aligned boundaries carry local continuity instead), and the state checkpoint is background only, never material to summarize into the stub. A slice consisting of recalled content is stubbed by code alone — a pointer line, no LLM call. A failed stub call degrades the same way: its slice gets a code-only pointer stub and the pass continues, making the state rewrite the only hard LLM dependency in a pass. ### The state checkpoint -One mutable working-memory document (at most one; zero before the first pass), positioned after all stubs and before the retained tail. Each pass rewrites it from the previous state plus this pass's staled content — O(previous + new), under the merge-don't-restate rule already in the summarization prompt — covering decisions, current state, constraints, and next steps. It carries its own footer and a size cap at the scale of today's summary. Stub calls receive the pass-start state as background, context only. +One mutable working-memory document (at most one; zero before the first pass), positioned after all stubs and before the retained tail. Each pass rewrites it from the previous state plus this pass's staled content — O(previous + new), under the merge-don't-restate rule already in the summarization prompt — covering decisions, current state, constraints, and next steps. It carries its own footer and a size cap at the scale of today's summary. An inflation guard bounds the whole pass: if the post-compaction size is not strictly below the pre-compaction size, nothing commits and the turn proceeds; the attempt defers until more stale history accumulates. The guard compares one metric on both sides — provider-reported usage (the counters PR #197 introduces), falling back to the character estimator on both sides. @@ -70,6 +70,7 @@ Specified during review, deferred until observation calls for them: - Periodic state refresh from chunk originals — on observed drift in the handoff probe. - `stateFallbackThreshold` (full-detail state prompt below a stub count) — on short-session regression. - Lazy registration of the recall tools — on measured context tax in never-compacting sessions. +- Amortized stub drafting at pre-step: as soon as stale-but-uncompacted content accumulates past `chunkTokens`, draft that chunk's stub at the next pre-step (a log-only draft event, written while the chunk's surrounding context is still live) and let the compaction pass commit drafts instead of summarizing in bulk — the deterministic, replay-exact equivalent of background compaction (the Claude Code session-memory pattern; OpenClaw demonstrates the synchronous semantics are identical). Trigger: observed pass latency, or stub-quality gains from drafting near-live proving out. - Split summarizer models; model-chosen chunk boundaries; cross-session recall; semantic search fallback — each behind its own evidence. ## Alternatives considered From 7089c88e4a13e719ad4c071156d531a4fe0e763d Mon Sep 17 00:00:00 2001 From: Ziya <199893125+ZiyaZhang@users.noreply.github.com> Date: Tue, 7 Jul 2026 21:07:06 -0700 Subject: [PATCH 3/4] docs(rfc): self-contained work references; add richer search query forms follow-up --- .../proposed/feature/2026-07-06-recallable-compaction.md | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md b/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md index e657887fe1..1de660ac31 100644 --- a/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md +++ b/docs/rfc/proposed/feature/2026-07-06-recallable-compaction.md @@ -28,7 +28,7 @@ A committed stub is never rewritten and never re-enters a later compaction regio One mutable working-memory document (at most one; zero before the first pass), positioned after all stubs and before the retained tail. Each pass rewrites it from the previous state plus this pass's staled content — O(previous + new), under the merge-don't-restate rule already in the summarization prompt — covering decisions, current state, constraints, and next steps. It carries its own footer and a size cap at the scale of today's summary. -An inflation guard bounds the whole pass: if the post-compaction size is not strictly below the pre-compaction size, nothing commits and the turn proceeds; the attempt defers until more stale history accumulates. The guard compares one metric on both sides — provider-reported usage (the counters PR #197 introduces), falling back to the character estimator on both sides. +An inflation guard bounds the whole pass: if the post-compaction size is not strictly below the pre-compaction size, nothing commits and the turn proceeds; the attempt defers until more stale history accumulates. The guard compares one metric on both sides — provider-reported usage from the request path, falling back to the character estimator on both sides. ### Pass execution @@ -56,8 +56,8 @@ The design ships as a new backend `dsh-compact-recallable` on the existing `ctx. ### Relation to in-flight work -- **Tool-result pruning (PR #113)**: its replacement nodes carry `sourceEventSeqs`; the same registry fold lists pruned results as recallable. Follow-up scope; neither blocks the other. -- **Provider-usage token pressure (PR #197)**: supplies the guard's accounting; the implementation stacks after it. +- **Tool-result pruning** (the in-flight pruning service): its replacement nodes carry `sourceEventSeqs`; the same registry fold lists pruned results as recallable. Follow-up scope; neither blocks the other. +- **Provider-usage token accounting** (the in-flight move of compaction pressure onto provider-reported usage): supplies the guard's accounting; the implementation stacks after it. - **"Query sessions" backlog item**: the cross-session generalization; this RFC scopes to the live session with tool names and rendering chosen so that work extends rather than collides. - **Training**: when to recall is a learned behavior. The deterministic footers and keyword anchors give training a stable target, and recall usage is fully visible in the session log for trajectory export; benchmark and RL design proceed with the post-training side. @@ -72,6 +72,7 @@ Specified during review, deferred until observation calls for them: - Lazy registration of the recall tools — on measured context tax in never-compacting sessions. - Amortized stub drafting at pre-step: as soon as stale-but-uncompacted content accumulates past `chunkTokens`, draft that chunk's stub at the next pre-step (a log-only draft event, written while the chunk's surrounding context is still live) and let the compaction pass commit drafts instead of summarizing in bulk — the deterministic, replay-exact equivalent of background compaction (the Claude Code session-memory pattern; OpenClaw demonstrates the synchronous semantics are identical). Trigger: observed pass latency, or stub-quality gains from drafting near-live proving out. - Split summarizer models; model-chosen chunk boundaries; cross-session recall; semantic search fallback — each behind its own evidence. +- Richer `history_search` query forms — regex, and structured queries over logged JSON tool results (sql/jq-style, or agent-authored queries against an indexed store) — on demand from observed search misses; literal matching ships first because the recall path stays a pure function of the log. ## Alternatives considered From 98ea306c5e41d54223b0a531c966a3bee3747b50 Mon Sep 17 00:00:00 2001 From: Yichen Jiang Date: Fri, 17 Jul 2026 10:31:44 +0800 Subject: [PATCH 4/4] fix(session-persistence): retire disposed coordinator state --- docs/event-producer-consumer.md | 2 +- ...18-shared-persistence-write-coordinator.md | 6 +- .../session-persistence/README.md | 2 + .../session-persistence/src/coordinator.ts | 90 +++-- .../tests/coordinator-contract.ts | 15 +- .../tests/persistence.spec.ts | 313 +++++++++++++++++- 6 files changed, 397 insertions(+), 31 deletions(-) diff --git a/docs/event-producer-consumer.md b/docs/event-producer-consumer.md index d8c924b24d..b6eb862884 100644 --- a/docs/event-producer-consumer.md +++ b/docs/event-producer-consumer.md @@ -26,7 +26,7 @@ This matrix shows which packages dispatch each harness-owned event and which pac | `fs/write-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:51`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`) | [`fs-policy`](../packages/fs/fs-policy) | | `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:39`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`invariants`](../packages/support/invariants), [`llm-replay`](../packages/support/llm-replay) | | `session/created` | `emit` | [`packages/core/session/src/index.ts:47`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`invariants`](../packages/support/invariants), [`jsonrpc`](../packages/ui/jsonrpc), [`session-persistence`](../packages/session-persistence/session-persistence) | -| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:57`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | - | +| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:57`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence`](../packages/session-persistence/session-persistence) | | `session/event` | `emit` | [`packages/core/session/src/index.ts:69`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/ui/acp), [`invariants`](../packages/support/invariants), [`jsonrpc`](../packages/ui/jsonrpc), [`session-persistence`](../packages/session-persistence/session-persistence), [`stdio`](../packages/ui/stdio) | | `session/flush` | `parallel` | [`packages/core/session/src/index.ts:79`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence`](../packages/session-persistence/session-persistence) | | `subagent/end` | `emit` | [`packages/subagent/subagent/src/index.ts:108`](../packages/subagent/subagent/src/index.ts) | [`subagent`](../packages/subagent/subagent) (`events.dispatch`) | [`hooks-claude`](../packages/hooks/hooks-claude), [`jsonrpc`](../packages/ui/jsonrpc) | diff --git a/docs/rfc/implemented/architecture/2026-06-18-shared-persistence-write-coordinator.md b/docs/rfc/implemented/architecture/2026-06-18-shared-persistence-write-coordinator.md index b1773b480d..fef6ad034c 100644 --- a/docs/rfc/implemented/architecture/2026-06-18-shared-persistence-write-coordinator.md +++ b/docs/rfc/implemented/architecture/2026-06-18-shared-persistence-write-coordinator.md @@ -12,6 +12,8 @@ Extract a backend-agnostic `PersistenceCoordinator` into `dsh-session-persistenc Composition, not inheritance. The coordinator is a concrete class the backend holds, not a base class the backend extends. The RFC's risk — "a coordinator must not make unusual backends fight an inheritance hierarchy" — is avoided: a backend exposes only the hooks; it cannot reach the coordinator's private orchestration state, and the public `SessionPersistence` service shape is unchanged, so a third-party backend MAY still implement the abstract service directly without the coordinator at all. +The coordinator retires each live session from its `session/disposed` notification: it waits for that exact Session object's initialization, serializes a final drain, and then removes the owned state, buffer, and init entries. Failed drains retain their buffers for backend teardown to retry. Settled per-id chain tails remove themselves only when they are still the current tail, so a completion cannot erase a newer operation for the same id. Backend teardown unregisters the write-path listeners before awaiting all admitted retirements, remaining buffers, and chains, then closes the backend. + ### The hook interface (`PersistenceBackend`) Six methods (five required + an optional lifecycle hook) — the only seam between the coordinator and storage: @@ -30,7 +32,7 @@ The single design choice that keeps the seam clean: the crash-repair "where is t ## Testing -The shared `runPersistenceContract` (public-API contract) keeps running for every backend. A new `runCoordinatorContract` (`tests/coordinator-contract.ts`) holds the write-path orchestration — adoption, HMR, collision, dispose-drain, crash-tail repair — and runs once per backend through a `CoordinatorFixture` (an in-memory reference + jsonl + sqlite). The per-backend specs shrank to storage mechanics only (JSONL: path safety, fsync rollback, bucket listing; SQLite: schema version, `scanRows`, transaction rollback). A through-coordinator torn-tail→load→`commitRepair` test per real backend (via a `corruptTail` fixture hook) keeps the coordinator's torn-marker repair branch covered under the 100% per-file gate — the contract crash test only produces synthetic closers, never a torn marker, so it could not reach that branch. +The shared `runPersistenceContract` (public-API contract) keeps running for every backend. `runCoordinatorContract` (`tests/coordinator-contract.ts`) holds the write-path orchestration — adoption, HMR, collision, session and backend disposal drains, and crash-tail repair — and runs once per backend through a `CoordinatorFixture` (an in-memory reference + jsonl + sqlite). Coordinator-specific tests pin retirement map cleanup, same-id chain-tail races, failed-drain retry, and close ordering. The per-backend specs retain storage mechanics only (JSONL: path safety, fsync rollback, bucket listing; SQLite: schema version, `scanRows`, transaction rollback). A through-coordinator torn-tail→load→`commitRepair` test per real backend (via a `corruptTail` fixture hook) keeps the coordinator's torn-marker repair branch covered under the 100% per-file gate — the contract crash test only produces synthetic closers, never a torn marker, so it could not reach that branch. ## Alternatives considered @@ -39,4 +41,4 @@ The shared `runPersistenceContract` (public-API contract) keeps running for ever ## Consequences -The coordinator adds one indirection and an opaque torn marker, but centralizes correctness-heavy orchestration previously duplicated by every backend. Its hook surface stays narrow: collision checks reuse `loadStored`, materialization stays atomic inside `appendBatch`, and listing bypasses the coordinator. New backends implement storage primitives rather than copy the event-buffer-flush lifecycle. +The coordinator adds one indirection, an opaque torn marker, and detached session-retirement tasks, but centralizes correctness-heavy orchestration previously duplicated by every backend. Session disposal remains an observe-only event, so the session owner does not await persistence retirement; the coordinator contains failures, preserves uncommitted buffers, and makes backend teardown the quiescence boundary. Its hook surface stays narrow: collision checks reuse `loadStored`, materialization stays atomic inside `appendBatch`, and listing bypasses the coordinator. New backends implement storage primitives rather than copy the event-buffer-flush lifecycle. diff --git a/packages/session-persistence/session-persistence/README.md b/packages/session-persistence/session-persistence/README.md index f09a21251b..bc901a38ed 100644 --- a/packages/session-persistence/session-persistence/README.md +++ b/packages/session-persistence/session-persistence/README.md @@ -24,6 +24,8 @@ The persisted unit IS the existing `SessionEvent` (event-sourced model — the l `PersistenceCoordinator` owns per-id state, write-behind buffers and serialization, the `session/event` → `session/flush` drain, lazy materialization, crash-tail repair, session adoption, and quiescent disposal. A first-party backend composes one, implements the small `PersistenceBackend` storage hook interface, and delegates its four public service methods. JSONL and SQLite therefore share lifecycle correctness while retaining different storage primitives; see the [coordinator RFC](../../../docs/rfc/implemented/architecture/2026-06-18-shared-persistence-write-coordinator.md). +When a live session emits `session/disposed`, the coordinator waits for its initialization, serializes a final buffer drain, then releases every map entry owned by that exact `Session` object. A failed final drain keeps the pending buffer for backend teardown to retry. Backend teardown stops event admission first, awaits all in-flight session retirements and remaining per-id operations, drains any retained buffers, and only then closes the storage handle. + The `PersistenceBackend` hooks (the only seam between the coordinator and storage): | Hook | Role | diff --git a/packages/session-persistence/session-persistence/src/coordinator.ts b/packages/session-persistence/session-persistence/src/coordinator.ts index 4a2fed8d34..686b9790d4 100644 --- a/packages/session-persistence/session-persistence/src/coordinator.ts +++ b/packages/session-persistence/session-persistence/src/coordinator.ts @@ -126,7 +126,8 @@ function seedCoversPrefix(seed: readonly SessionEvent[], prefix: readonly Sessio * * All per-id operations are serialized (a per-id promise chain) so concurrent * flushes / a flush racing a load never interleave storage writes. The - * constructor installs the write-path listeners and the dispose effect. + * constructor installs the write-path listeners, per-session retirement, and + * the backend dispose effect. * * @typeParam TornMarker - the backend's opaque torn-tail repair token. */ @@ -146,6 +147,8 @@ export class PersistenceCoordinator { * observation boundary; callers do not inspect this bookkeeping directly. */ private inits = new Map>() + /** Final drains started by fire-and-forget session disposal notifications. */ + private retirements = new Set>() constructor(private ctx: Context, private backend: PersistenceBackend) { this.installWritePath() @@ -267,7 +270,13 @@ export class PersistenceCoordinator { const next = prior.then(op, op) // Keep the chain alive but swallow this op's rejection for the NEXT waiter // (the caller still sees the real rejection via `next`). - this.chains.set(id, next.then(() => undefined, () => undefined)) + const tail = next.then(() => undefined, () => undefined) + this.chains.set(id, tail) + // Settled tails carry no serialization value. Delete only the exact tail + // installed above: a later operation may already have replaced it. + void tail.then(() => { + if (this.chains.get(id) === tail) this.chains.delete(id) + }) return next } @@ -293,27 +302,12 @@ export class PersistenceCoordinator { private installWritePath(): void { const ctx = this.ctx - // Capture the header on creation; persist a fork's seed once. Record the init - // promise so flush/dispose can await it (onCreated is async). - ctx.on('session/created', (session) => { void this.initFor(session) }) - - // Session emits an owned frozen event. Keep a persistence-owned copy anyway - // so the write-behind queue owns exactly the record it will flush rather than - // retaining a product-layer record by identity. Serializability is guaranteed - // at the source, so structuredClone is safe. - ctx.on('session/event', (session, event) => { - let buffer = this.buffers.get(session) - if (!buffer) this.buffers.set(session, buffer = []) - buffer.push(structuredClone(event)) - }) - - // Drain to the backend at the durability checkpoint. - ctx.on('session/flush', session => this.flush(session)) - - // Dispose must reach quiescence: await every init + final drain BEFORE - // returning, then close the backend's own resources (AFTER the drain), so no - // write lands after teardown and a close failure never MASKS a drain error. + // Register the disposer BEFORE the listeners. Cordis tears effects down in + // reverse registration order, so event admission closes before this final + // drain reaches quiescence and closes the backend. ctx.effect(() => async () => { + await this.awaitRetirements() + let disposeError: unknown try { const errors = [ @@ -341,11 +335,63 @@ export class PersistenceCoordinator { } }, `${this.backend.name} write path`) + // Capture the header on creation; persist a fork's seed once. Record the init + // promise so flush/dispose can await it (onCreated is async). + ctx.on('session/created', (session) => { void this.initFor(session) }) + + // Session emits an owned frozen event. Keep a persistence-owned copy anyway + // so the write-behind queue owns exactly the record it will flush rather than + // retaining a product-layer record by identity. Serializability is guaranteed + // at the source, so structuredClone is safe. + ctx.on('session/event', (session, event) => { + let buffer = this.buffers.get(session) + if (!buffer) this.buffers.set(session, buffer = []) + buffer.push(structuredClone(event)) + }) + + // Drain to the backend at the durability checkpoint. + ctx.on('session/flush', session => this.flush(session)) + + // Session disposal is observe-only, so the coordinator observes the + // detached task itself and backend teardown awaits quiescence. + ctx.on('session/disposed', (session) => { this.retire(session) }) + // HMR: a hot reload does not replay session/created, so seed existing live // sessions (mirrors dsh-invariants). for (const session of ctx.sessions.list()) void this.initFor(session) } + /** Start, observe, and track one disposed session's final drain. */ + private retire(session: Session): void { + const task = this.retireCore(session) + this.retirements.add(task) + const settled = (): void => { this.retirements.delete(task) } + void task.then(settled, (error: unknown) => { + settled() + this.ctx.logger.warn(`${this.backend.name}: session "${session.id}" retirement failed: ${String(error)}`) + }) + } + + /** Drain and release state owned by one exact disposed Session lifecycle. */ + private async retireCore(session: Session): Promise { + await this.inits.get(session) + + const id = session.header.id + await this.serialize(id, async () => { + await this.drain(session) + this.buffers.delete(session) + this.inits.delete(session) + if (this.states.get(id)?.owner === session) this.states.delete(id) + }) + } + + /** Await every retirement admitted before listener teardown. */ + private async awaitRetirements(): Promise { + while (this.retirements.size > 0) { + await Promise.allSettled([...this.retirements]) + } + } + /** Start (once) the async init for a session and remember its promise. */ private initFor(session: Session): Promise { const existing = this.inits.get(session) diff --git a/packages/session-persistence/session-persistence/tests/coordinator-contract.ts b/packages/session-persistence/session-persistence/tests/coordinator-contract.ts index f6692b9175..defcaefb6d 100644 --- a/packages/session-persistence/session-persistence/tests/coordinator-contract.ts +++ b/packages/session-persistence/session-persistence/tests/coordinator-contract.ts @@ -9,7 +9,7 @@ * @module @deepseek-ai/dsh-session-persistence/tests/coordinator-contract */ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { Context, type Fiber } from 'cordis' import SessionStore, { SESSION_FORMAT_VERSION, SessionId } from '@deepseek-ai/dsh-session' import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' @@ -401,7 +401,7 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise< } }) - it('does NOT reclaim an id whose abandoned owner still has buffered (unflushed) events', async () => { + it('session disposal drains buffered events before retiring ownership', async () => { const fix = await makeFixture() const { ctx, fiber } = await freshCtx(fix) try { @@ -413,13 +413,20 @@ export function runCoordinatorContract(name: string, makeFixture: () => Promise< // Append a turn but do NOT flush — events sit in the write-behind buffer. first.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) first.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) - await firstFiber.dispose() // disposed before flush; not materialized, buffer pending + await firstFiber.dispose() + + // Disposal is an observe-only notification. Poll storage rather than + // assuming the owning fiber awaits the coordinator's detached drain. + await vi.waitFor(async () => { + expect((await ctx.sessionPersistence.list()).map(meta => meta.id)).toContain(SessionId('buffered')) + }) + expect((await ctx.sessionPersistence.load(SessionId('buffered'))).events.map(event => event.seq)).toEqual([0, 1]) let reuse!: Session await ctx.plugin(Object.assign((inner: Context) => { reuse = inner.sessions.create(SessionId('buffered'), { meta: { cwd: WORK } }) }, { inject: ['sessions'] })) - await expect(ctx.sessions.flush(reuse)).rejects.toThrow(/already bound to a different live session/) + await expect(ctx.sessions.flush(reuse)).rejects.toThrow(/persisted log|id collision/) } finally { await fiber.dispose() await fix.cleanup() diff --git a/packages/session-persistence/session-persistence/tests/persistence.spec.ts b/packages/session-persistence/session-persistence/tests/persistence.spec.ts index a937a39da7..bd7403bbd2 100644 --- a/packages/session-persistence/session-persistence/tests/persistence.spec.ts +++ b/packages/session-persistence/session-persistence/tests/persistence.spec.ts @@ -1,7 +1,7 @@ -import { describe, expect, it } from 'vitest' +import { describe, expect, it, vi } from 'vitest' import { Context } from 'cordis' import SessionStore, { SessionId, isJsonValue } from '@deepseek-ai/dsh-session' -import type { SessionEvent, SessionHeader } from '@deepseek-ai/dsh-session' +import type { Session, SessionEvent, SessionHeader } from '@deepseek-ai/dsh-session' import { SessionPersistence, PersistenceCoordinator, type PersistenceBackend, type StoredPrefix, @@ -15,6 +15,15 @@ type MemoryStore = Map /** Optional plugin config: an EXTERNAL store shared across backend instances. */ interface MemoryConfig { store?: MemoryStore } +/** Test-only view of the coordinator containers whose retirement is the contract under test. */ +interface CoordinatorInternals { + states: Map + buffers: Map + chains: Map + inits: Map + retirements: Set> +} + /** * Reference {@link PersistenceCoordinator} vehicle and abstract-service coverage, backed by a * dependency-free map with atomic writes and no torn-tail marker. Supplying the map lets multiple @@ -97,6 +106,49 @@ class MemoryPersistence extends SessionPersistence implements PersistenceBackend } } +/** Controllable storage primitive for serialization and retirement failure tests. */ +class ControlledBackend implements PersistenceBackend { + readonly name = 'session-persistence-controlled' + readonly store: MemoryStore = new Map() + readonly lifecycle: string[] = [] + appendAttempts = 0 + loadAttempts = 0 + beforeAppend?: (attempt: number) => Promise + beforeLoadStored?: (attempt: number) => Promise + + async loadStored(id: SessionId): Promise | undefined> { + await this.beforeLoadStored?.(++this.loadAttempts) + const entry = this.store.get(id) + if (entry === undefined) return undefined + return { meta: structuredClone(entry.meta), events: structuredClone(entry.events) } + } + + loadLive(id: SessionId, _cwd: string | undefined): Promise | undefined> { + return this.loadStored(id) + } + + async appendBatch(m: SessionHeader, events: readonly SessionEvent[], _isMaterialized: boolean): Promise { + const attempt = ++this.appendAttempts + await this.beforeAppend?.(attempt) + const entry = this.store.get(m.id) + if (entry === undefined) { + this.store.set(m.id, { meta: structuredClone(m), events: structuredClone(events) as SessionEvent[] }) + } else { + entry.events.push(...structuredClone(events) as SessionEvent[]) + } + } + + async commitRepair(_m: SessionHeader, _tornMarker: undefined, _closers: readonly SessionEvent[]): Promise {} + + async list(): Promise { + return [...this.store.values()].map(entry => structuredClone(entry.meta)) + } + + async close(): Promise { + this.lifecycle.push('close') + } +} + // Run the shared contract against the in-memory backend. runPersistenceContract('memory', async () => { const ctx = new Context() @@ -118,6 +170,230 @@ runCoordinatorContract('memory', async (): Promise => { } }) +describe('PersistenceCoordinator retirement', () => { + it('a retiring unmaterialized owner without buffered events releases its id', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + const backend = new ControlledBackend() + let coordinator!: PersistenceCoordinator + const backendFiber = await ctx.plugin(Object.assign((inner: Context) => { + coordinator = new PersistenceCoordinator(inner, backend) + }, { inject: ['sessions'] })) + const loadGate = Promise.withResolvers() + + try { + const id = SessionId('retiring-lazy-owner') + let first!: Session + const firstFiber = await ctx.plugin(Object.assign((inner: Context) => { + first = inner.sessions.create(id) + }, { inject: ['sessions'] })) + await ctx.sessions.flush(first) + + const baselineLoads = backend.loadAttempts + backend.beforeLoadStored = async () => { await loadGate.promise } + const blockingLoad = coordinator.load(id) + await vi.waitFor(() => { expect(backend.loadAttempts).toBe(baselineLoads + 1) }) + await firstFiber.dispose() + + let reuse!: Session + await ctx.plugin(Object.assign((inner: Context) => { + reuse = inner.sessions.create(id) + }, { inject: ['sessions'] })) + await vi.waitFor(() => { expect(backend.loadAttempts).toBe(baselineLoads + 2) }) + + loadGate.resolve(true) + await expect(blockingLoad).rejects.toThrow(/not found/) + await expect(ctx.sessions.flush(reuse)).resolves.toBeUndefined() + } finally { + loadGate.resolve(true) + await backendFiber.dispose() + await ctx.fiber.dispose() + } + }) + + it('a retiring owner with buffered events still rejects same-id reuse', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + const backend = new ControlledBackend() + let coordinator!: PersistenceCoordinator + const backendFiber = await ctx.plugin(Object.assign((inner: Context) => { + coordinator = new PersistenceCoordinator(inner, backend) + }, { inject: ['sessions'] })) + const loadGate = Promise.withResolvers() + + try { + const id = SessionId('retiring-buffered-owner') + let first!: Session + const firstFiber = await ctx.plugin(Object.assign((inner: Context) => { + first = inner.sessions.create(id) + }, { inject: ['sessions'] })) + await ctx.sessions.flush(first) + first.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) + first.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) + + const baselineLoads = backend.loadAttempts + backend.beforeLoadStored = async () => { await loadGate.promise } + const blockingLoad = coordinator.load(id) + await vi.waitFor(() => { expect(backend.loadAttempts).toBe(baselineLoads + 1) }) + await firstFiber.dispose() + + let reuse!: Session + await ctx.plugin(Object.assign((inner: Context) => { + reuse = inner.sessions.create(id) + }, { inject: ['sessions'] })) + await expect(ctx.sessions.flush(reuse)).rejects.toThrow(/bound to a different live session/) + + loadGate.resolve(true) + await expect(blockingLoad).rejects.toThrow(/not found/) + await vi.waitFor(() => { + expect(backend.store.get(id)?.events.map(event => event.seq)).toEqual([0, 1]) + }) + } finally { + loadGate.resolve(true) + await backendFiber.dispose() + await ctx.fiber.dispose() + } + }) + + it('a settled chain tail cannot delete a newer operation for the same id', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + const backend = new ControlledBackend() + let coordinator!: PersistenceCoordinator + const fiber = await ctx.plugin(Object.assign((inner: Context) => { + coordinator = new PersistenceCoordinator(inner, backend) + }, { inject: ['sessions'] })) + const internals = coordinator as unknown as CoordinatorInternals + const first = Promise.withResolvers() + const second = Promise.withResolvers() + backend.beforeAppend = async (attempt) => { + if (attempt === 1) await first.promise + if (attempt === 2) await second.promise + } + + try { + const id = SessionId('chain-tail') + await coordinator.create(meta(id)) + const firstAppend = coordinator.append(id, [{ + type: 'turn/start', + seq: 0, + time: 1, + data: { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }, + }]) + const secondAppend = coordinator.append(id, [{ + type: 'turn/end', + seq: 1, + time: 2, + data: { turn: 1, reason: { kind: 'completed' } }, + }]) + + await vi.waitFor(() => { expect(backend.appendAttempts).toBe(1) }) + first.resolve(true) + await vi.waitFor(() => { expect(backend.appendAttempts).toBe(2) }) + expect(internals.chains.size).toBe(1) + second.resolve(true) + await Promise.all([firstAppend, secondAppend]) + await vi.waitFor(() => { expect(internals.chains.size).toBe(0) }) + expect(backend.store.get(id)?.events.map(event => event.seq)).toEqual([0, 1]) + } finally { + first.resolve(true) + second.resolve(true) + await fiber.dispose() + await ctx.fiber.dispose() + } + }) + + it('backend teardown retries a failed session retirement before close', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + const backend = new ControlledBackend() + let coordinator!: PersistenceCoordinator + const backendFiber = await ctx.plugin(Object.assign((inner: Context) => { + coordinator = new PersistenceCoordinator(inner, backend) + }, { inject: ['sessions'] })) + const internals = coordinator as unknown as CoordinatorInternals + backend.beforeAppend = async (attempt) => { + if (attempt === 1) { + backend.lifecycle.push('append-failed') + throw new Error('transient append failure') + } + backend.lifecycle.push('append-committed') + } + + try { + let session!: Session + const sessionFiber = await ctx.plugin(Object.assign((inner: Context) => { + session = inner.sessions.create(SessionId('retry-retirement')) + }, { inject: ['sessions'] })) + session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) + session.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) + await sessionFiber.dispose() + + await vi.waitFor(() => { + expect(backend.appendAttempts).toBe(1) + expect(internals.retirements.size).toBe(0) + }) + expect([...internals.buffers.values()]).toEqual([expect.arrayContaining([ + expect.objectContaining({ seq: 0 }), + expect.objectContaining({ seq: 1 }), + ])]) + + await backendFiber.dispose() + expect(backend.store.get(SessionId('retry-retirement'))?.events.map(event => event.seq)).toEqual([0, 1]) + expect(backend.lifecycle).toEqual(['append-failed', 'append-committed', 'close']) + } finally { + await backendFiber.dispose() + await ctx.fiber.dispose() + } + }) + + it('backend teardown waits for an in-flight session retirement before close', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + const backend = new ControlledBackend() + let coordinator!: PersistenceCoordinator + const backendFiber = await ctx.plugin(Object.assign((inner: Context) => { + coordinator = new PersistenceCoordinator(inner, backend) + }, { inject: ['sessions'] })) + const internals = coordinator as unknown as CoordinatorInternals + const appendGate = Promise.withResolvers() + backend.beforeAppend = async () => { + backend.lifecycle.push('append-started') + await appendGate.promise + backend.lifecycle.push('append-committed') + } + + try { + let session!: Session + const sessionFiber = await ctx.plugin(Object.assign((inner: Context) => { + session = inner.sessions.create(SessionId('inflight-retirement')) + }, { inject: ['sessions'] })) + session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) + session.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) + await sessionFiber.dispose() + await vi.waitFor(() => { + expect(backend.appendAttempts).toBe(1) + expect(internals.retirements.size).toBe(1) + }) + + let disposed = false + const teardown = backendFiber.dispose().then(() => { disposed = true }) + await Promise.resolve() + expect(disposed).toBe(false) + expect(backend.lifecycle).toEqual(['append-started']) + + appendGate.resolve(true) + await teardown + expect(backend.store.get(SessionId('inflight-retirement'))?.events.map(event => event.seq)).toEqual([0, 1]) + expect(backend.lifecycle).toEqual(['append-started', 'append-committed', 'close']) + } finally { + appendGate.resolve(true) + await backendFiber.dispose() + await ctx.fiber.dispose() + } + }) +}) + describe('SessionPersistence service registration', () => { it('registers as ctx.sessionPersistence and is removed on fiber dispose (HMR safety)', async () => { const ctx = new Context() @@ -151,4 +427,37 @@ describe('SessionPersistence service registration', () => { .rejects.toThrow('session metadata must be losslessly JSON-serializable') await fiber.dispose() }) + + it('retires all coordinator bookkeeping for disposed sessions', async () => { + const ctx = new Context() + await ctx.plugin(SessionStore) + const fiber = await ctx.plugin(MemoryPersistence) + const { coordinator } = ctx.sessionPersistence as unknown as { coordinator: CoordinatorInternals } + + try { + for (let index = 0; index < 3; index += 1) { + let session!: Session + const sessionFiber = await ctx.plugin(Object.assign((inner: Context) => { + session = inner.sessions.create(SessionId(`disposed-${index}`)) + }, { inject: ['sessions'] })) + session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) + session.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) + await ctx.sessions.flush(session) + await sessionFiber.dispose() + } + + await vi.waitFor(() => { + expect(ctx.sessions.list()).toHaveLength(0) + expect({ + states: coordinator.states.size, + buffers: coordinator.buffers.size, + chains: coordinator.chains.size, + inits: coordinator.inits.size, + retirements: coordinator.retirements.size, + }).toEqual({ states: 0, buffers: 0, chains: 0, inits: 0, retirements: 0 }) + }) + } finally { + await fiber.dispose() + } + }) })