From 171bae5c20ee23abfafaf7a244e13ce1a2692649 Mon Sep 17 00:00:00 2001 From: Hypatia May Date: Thu, 16 Jul 2026 18:28:31 +0800 Subject: [PATCH] fix(compact): harden pruning integration (round 2) --- docs/architecture.md | 4 +- docs/core-data-structures/compaction.md | 2 +- ...n-pressure-and-overflow-recovery.i18n.yaml | 4 +- ...mpaction-pressure-and-overflow-recovery.md | 6 +- ...ction-pressure-and-overflow-recovery.zh.md | 6 +- packages/compact/compact-basic/README.md | 4 +- packages/compact/compact-basic/src/index.ts | 19 +++- .../compact-basic/tests/compact-basic.spec.ts | 45 +++++++++ packages/support/invariants/README.md | 2 +- packages/support/invariants/src/index.ts | 73 ++++++++++---- .../invariants/tests/invariants.spec.ts | 96 +++++++++++++++---- packages/ui/acp/README.md | 2 +- packages/ui/acp/acp-feature-support.md | 2 +- packages/ui/acp/src/index.ts | 7 +- packages/ui/acp/tests/load.spec.ts | 47 +++++++++ packages/ui/acp/tests/stream-update.spec.ts | 74 +++++++++++++- packages/ui/stdio/README.md | 2 +- packages/ui/stdio/src/index.ts | 4 + packages/ui/stdio/tests/stdio.spec.ts | 37 +++++++ 19 files changed, 372 insertions(+), 64 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index d2289ff731..ff45d38389 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -107,11 +107,11 @@ Each step renders one prompt assembly. Plugins contribute ordered sections, tool Post-tool context follows all results, preserving call/result adjacency. Steering drains before `agent/post-step`, which observes durable output, results, context, and steering while the step signal remains open. Leftover steering becomes next-turn input. `agent/turn-stop` is terminal through close and flush: later steering is discarded, while ordinary queued prompts survive. -When loaded, `dsh-compact-basic` consumes that post-step checkpoint for `ctx.tokenMeter` pressure under the actual routed header. Once pressure or canonical context overflow qualifies, it runs optional `ctx.toolResultPrune` rewriting before summary selection and remeasures the replayed surface. Overflow recovery authorizes retry after either pruning or tool-balanced summary compaction advances `surface.replaceGeneration`. The same turn signal owns both paths. +When loaded, `dsh-compact-basic` consumes that post-step checkpoint for `ctx.tokenMeter` pressure under the actual routed header. Once pressure or canonical context overflow qualifies, it runs optional `ctx.toolResultPrune` rewriting before summary selection and remeasures the replayed surface. Overflow recovery authorizes retry after either pruning or tool-balanced summary compaction advances `surface.replaceGeneration`, including when later summary work fails after a prune. The same turn signal owns both paths, and cancellation still wins. ### Failure Boundaries -The turn is the containment boundary. `LlmService` preserves and privately tags errors from final adapter selection, dispatch, and iteration. Those errors and terminal in-band error/aborted finishes close the failed step before `agent/request-error`; retry reconstructs the next numbered step from the log, while decline or failed recovery preserves the provider error. Attempts count consecutive failures and reset after success. +The turn is the containment boundary. `LlmService` preserves and privately tags errors from final adapter selection, dispatch, and iteration. Those errors and terminal in-band error/aborted finishes close the failed step before `agent/request-error`; retry reconstructs the next numbered step from the log, while decline or recovery failure before any replacement preserves the provider error. Attempts count consecutive failures and reset after success. Prompt, middleware, result, tool, post-step, and continuation failures remain ordinary `agent/error` failures. Cancellation and disposal beat recovery. Durable undispatched tool calls receive synthetic `ABORTED` results, preventing dangling replay. `cancel()` clears queues and aborts active work; disposal awaits quiescence before unregistering. diff --git a/docs/core-data-structures/compaction.md b/docs/core-data-structures/compaction.md index 3a008a0b95..e3774b87cd 100644 --- a/docs/core-data-structures/compaction.md +++ b/docs/core-data-structures/compaction.md @@ -58,6 +58,6 @@ export type CompactionTrigger = 'pressure' | 'context-overflow' `CompactService` exposes `compactIfNeeded(agent, trigger, signal)` for automatic `pressure` or `context-overflow` policy, returning `null` when no safe work exists, and `compactRegion(...)` for an explicit inclusive surface range. Implementations must forward the supplied signal to summarization. The seam owns no pricing API: the singleton [`ctx.tokenMeter`](token-meter.md) directly owns estimation and replay, while `dsh-compact-basic` owns retention, event sequencing, routed summarization calls, and their configuration. -Pressure compaction runs at serial `agent/post-step`, after successful assistant output, tool results, buffered context, and steering are durable but before `step/end`. Once pressure or canonical overflow qualifies, compact-basic invokes optional [`ctx.toolResultPrune`](../../packages/compact/tool-result-prune/README.md) before range selection, remeasures through `ctx.tokenMeter`, and can advance the surface without a summary. Failed-request recovery runs through `agent/request-error` after the failed step closes and authorizes a fresh numbered-step retry only when the surface replacement generation advances. Region boundaries preserve tool-call/result pairing but not whole turns, allowing early closed steps of one oversized turn to compact. `dsh-compact-basic` owns thresholds, retained-tail policy, overflow caps, and failure handling. +Pressure compaction runs at serial `agent/post-step`, after successful assistant output, tool results, buffered context, and steering are durable but before `step/end`. Once pressure or canonical overflow qualifies, compact-basic invokes optional [`ctx.toolResultPrune`](../../packages/compact/tool-result-prune/README.md) before range selection, remeasures through `ctx.tokenMeter`, and can advance the surface without a summary. Failed-request recovery runs through `agent/request-error` after the failed step closes and authorizes a fresh numbered-step retry only when the surface replacement generation advances, even if later summary work throws after pruning; cancellation still wins. Region boundaries preserve tool-call/result pairing but not whole turns, allowing early closed steps of one oversized turn to compact. `dsh-compact-basic` owns thresholds, retained-tail policy, overflow caps, and failure handling. The seam exports `toolPairingBalancedBefore(session, node)` and `toolPairingBalancedAfter(session, node)` for those edge checks. Both validate current surface membership, reject stale or missing seqs and orphan results, and ignore a caller-retained `node.next`; the [package contract](../../packages/compact/compact/README.md#tool-pairing-boundaries) owns their cache semantics. diff --git a/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml b/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml index bd13a336f1..c8eef0f1c9 100644 --- a/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml +++ b/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write -2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md: 99dc7625b8e185464d7a8a1ea8eda5baf0674df7 -2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md: 4f84d4435341005c058e32582e6d26b2a9f29bc1 +2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md: deedb81f8cf75ab80b70e2ef3148ba76d1278886 +2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md: 51d28fa243f522eedcac29002670575889d48369 diff --git a/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md b/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md index 99dc7625b8..deedb81f8c 100644 --- a/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md +++ b/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.md @@ -34,9 +34,9 @@ If cancellation lands after assistant tool calls are durable but before all call For `pressure`, compact-basic applies the service-wide threshold and retained-tail policy to one unified `ctx.tokenMeter.measure()` result. Below pressure it returns without pruning. Once pressure qualifies, optional `ctx.toolResultPrune` rewrites oversized current results and compact-basic remeasures through the same meter; safe pressure skips the model call, while remaining pressure selects and summarizes from the pruned surface. The same singleton meter owns range pricing, provenance, shadowed token counts, and non-shrinking-summary rejection. The common defaults remain threshold ratio `0.8`, retained history `floor(contextWindow × 0.16)`, summarization model `''`, `maxTokens: 8192`, `compactionRetries: 1`, and `auto: true`. -For canonical overflow, compact-basic bypasses scalar pressure and the normal retained-token budget. It prunes first, then chooses the maximal tool-balanced head range while leaving the newest indivisible unit and attempts one shrinking summary compaction under the same signal when a range exists. The automatic listener snapshots `session.surface.replaceGeneration` and returns `{ action: 'retry' }` whenever pruning or summarization increases it. A backend returning a result without replacement cannot authorize retry, while pruning-only progress can authorize a retry without a `CompactionResult`. +For canonical overflow, compact-basic bypasses scalar pressure and the normal retained-token budget. It prunes first, then chooses the maximal tool-balanced head range while leaving the newest indivisible unit and attempts one shrinking summary compaction under the same signal when a range exists. The automatic listener snapshots `session.surface.replaceGeneration` and returns `{ action: 'retry' }` whenever pruning or summarization increases it. This remains true when pruning lands before later summary work throws; cancellation still wins. A backend returning a result without replacement cannot authorize retry, while pruning-only progress can authorize a retry without a `CompactionResult`. -`maxOverflowRetries` is optional and defaults to `1`; `0` disables overflow recovery without disabling pressure. `auto: false` registers neither automatic listener. Noncanonical errors, exhausted attempts, an already-aborted signal, a missing routed model, no safe range, no generation change, and recovery throws all delegate to the next listener. With no later recovery, the loop reports the original provider error object and code. Cancellation or disposal remains authoritative even if recovery work completes concurrently. +`maxOverflowRetries` is optional and defaults to `1`; `0` disables overflow recovery without disabling pressure. `auto: false` registers neither automatic listener. Noncanonical errors, exhausted attempts, an already-aborted signal, a missing routed model, no safe range, no generation change, and recovery throws before any replacement all delegate to the next listener. With no later recovery, the loop reports the original provider error object and code. A recovery throw after generation advances authorizes retry from durable progress; cancellation or disposal remains authoritative even if recovery work completes concurrently. The default summarizer still resolves explicit configuration, then the latest logged route, then agent options. Because direct `llm/stream` middleware may reroute that auxiliary call, `compact/summary.model` records the final mutable `GenerateOptions.model` observed after dispatch rather than the pre-waterfall candidate. @@ -44,7 +44,7 @@ The default summarizer still resolves explicit configuration, then the latest lo Lifecycle tests pin post-step ordering after durable tool/context/steering work, content-less and max-token successes, final-adapter dispatch/iterator/in-band boundaries, retry numbering, attempt reset, cancellation, disposal, synthetic tool results, and original error identity. -Compact tests pin low-friction service-wide defaults, actual routed-model selection, unlisted-model measurement, unified pressure-and-retention decisions, pressure-gated pruning, pruning-only relief, summarization from pruned input, optional-plugin fallback, pruning-only and summarized overflow recovery, newest tool-pair retention, non-shrinking rejection, generation proof, caps, disabled listeners, single downstream delegation, and auxiliary summary routing provenance. Real-loop composition covers both thrown and in-band overflow: the failed step closes, compaction lands between attempts, and the next numbered request is reconstructed from the replacement surface. +Compact tests pin low-friction service-wide defaults, actual routed-model selection, unlisted-model measurement, unified pressure-and-retention decisions, pressure-gated pruning, pruning-only relief, summarization from pruned input, optional-plugin fallback, pruning-only and summarized overflow recovery, prune-then-summary-failure progress, newest tool-pair retention, non-shrinking rejection, generation proof, caps, disabled listeners, single downstream delegation, and auxiliary summary routing provenance. Real-loop composition covers both thrown and in-band overflow: the failed step closes, compaction lands between attempts, and the next numbered request is reconstructed from the replacement surface. ## Alternatives considered diff --git a/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md b/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md index 4f84d44353..51d28fa243 100644 --- a/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md +++ b/docs/rfc/implemented/architecture/2026-07-10-after-call-compaction-pressure-and-overflow-recovery.zh.md @@ -34,9 +34,9 @@ Status: implemented 对于 `pressure`,compact-basic 把服务级阈值与保留尾部策略应用到一次统一的 `ctx.tokenMeter.measure()` 结果。低于压力时直接返回,不执行剪枝。压力达到条件后,可选的 `ctx.toolResultPrune` 会改写当前表层中过大的工具结果,compact-basic 再通过同一个 meter 重新计量;若压力恢复安全则跳过模型调用,否则从已剪枝表层选择范围并生成摘要。范围定价、来源、被遮蔽 token 数与非缩小摘要拒绝也由同一个单例 meter 完成。通用默认值保持为阈值比例 `0.8`、保留历史 `floor(contextWindow × 0.16)`、摘要模型 `''`、`maxTokens: 8192`、`compactionRetries: 1` 与 `auto: true`。 -对于规范化溢出,compact-basic 绕过标量压力与普通保留 token 预算。它先执行剪枝,再在保留最新不可分割单元的同时选择最大的工具配对平衡头部范围;存在范围时,才在同一 signal 下尝试一次缩小摘要压缩。自动监听器先记录 `session.surface.replaceGeneration`,剪枝或摘要让 generation 增加时就返回 `{ action: 'retry' }`。后端若只返回结果但没有替换表层,不能授权重试;只有剪枝取得进展时,即使没有 `CompactionResult` 也可以授权重试。 +对于规范化溢出,compact-basic 绕过标量压力与普通保留 token 预算。它先执行剪枝,再在保留最新不可分割单元的同时选择最大的工具配对平衡头部范围;存在范围时,才在同一 signal 下尝试一次缩小摘要压缩。自动监听器先记录 `session.surface.replaceGeneration`,剪枝或摘要让 generation 增加时就返回 `{ action: 'retry' }`。即使剪枝先落盘而后续摘要工作抛错,这条规则仍然成立;取消依然优先。后端若只返回结果但没有替换表层,不能授权重试;只有剪枝取得进展时,即使没有 `CompactionResult` 也可以授权重试。 -`maxOverflowRetries` 可选且默认为 `1`;`0` 只禁用溢出恢复,不会禁用压力检查。`auto: false` 不注册任何自动监听器。非规范化错误、尝试耗尽、已经中止的 signal、缺失路由模型、没有安全范围、generation 未变化,以及恢复抛错都会委托给下一个监听器。若没有后续恢复,循环报告原始提供方错误对象与代码。即使恢复工作并发完成,取消或销毁仍具有最终优先级。 +`maxOverflowRetries` 可选且默认为 `1`;`0` 只禁用溢出恢复,不会禁用压力检查。`auto: false` 不注册任何自动监听器。非规范化错误、尝试耗尽、已经中止的 signal、缺失路由模型、没有安全范围、generation 未变化,以及在任何替换之前恢复抛错,都会委托给下一个监听器。若没有后续恢复,循环报告原始提供方错误对象与代码。generation 增加后的恢复抛错会基于持久进展授权重试;即使恢复工作并发完成,取消或销毁仍具有最终优先级。 默认摘要器仍依次解析显式配置、最近记录的路由与 agent options。因为直接 `llm/stream` 中间件可以重新路由该辅助调用,`compact/summary.model` 记录分发后最终可变的 `GenerateOptions.model`,而不是 waterfall 之前的候选值。 @@ -44,7 +44,7 @@ Status: implemented 生命周期测试固定 post-step 位于持久工具、上下文与 steering 工作之后,覆盖无内容与达到 token 上限的成功、最终适配器分发/迭代器/带内边界、重试编号、尝试重置、取消、销毁、合成工具结果与原始错误身份。 -压缩测试固定低摩擦服务级默认值、实际路由模型选择、未列出模型计量、统一压力与保留决策、压力门控剪枝、剪枝独立解除压力、从已剪枝输入生成摘要、可选插件回退、仅剪枝与剪枝后摘要两类溢出恢复、最新工具配对保留、非缩小拒绝、generation 证明、上限、禁用监听器、单次下游委托与辅助摘要路由来源。真实循环组合同时覆盖抛出式和带内溢出:失败 step 关闭,压缩落在两次尝试之间,下一个编号请求从替换表层重建。 +压缩测试固定低摩擦服务级默认值、实际路由模型选择、未列出模型计量、统一压力与保留决策、压力门控剪枝、剪枝独立解除压力、从已剪枝输入生成摘要、可选插件回退、仅剪枝与剪枝后摘要两类溢出恢复、剪枝后摘要失败的持久进展、最新工具配对保留、非缩小拒绝、generation 证明、上限、禁用监听器、单次下游委托与辅助摘要路由来源。真实循环组合同时覆盖抛出式和带内溢出:失败 step 关闭,压缩落在两次尝试之间,下一个编号请求从替换表层重建。 ## 考虑过的替代方案 diff --git a/packages/compact/compact-basic/README.md b/packages/compact/compact-basic/README.md index 75be39aa1b..7acc1d0351 100644 --- a/packages/compact/compact-basic/README.md +++ b/packages/compact/compact-basic/README.md @@ -15,8 +15,8 @@ This backend owns the compaction policy: - **Summarization** — a direct `llm/stream` call uses the configured model and cap without running the loop-only `agent/request` seam. The input transcript preserves non-text blocks as tagged placeholders; only returned text enters the checkpoint, excluding reasoning and tool calls that would leak private reasoning or create an orphaned call. - **Framing** — the replacement user message marks established checkpoint context with `` tags. The raw summary remains on the provenance event, and later automatic cycles merge the prior checkpoint. - **Lifecycle** — `compactRegion()` requires its agent to own the exact target session and rejects mismatch before resolution or mutation; a valid call records its start, summary, replacement, and end. The serial `agent/post-step` listener checks pressure after successful output and tool work are durable but before `step/end`. Canonical provider overflow is handled through `agent/request-error` after the failed step closes. -- **Overflow recovery** — below-threshold overflow bypasses normal retention and first prunes, then attempts one maximal balanced head reduction while leaving the newest indivisible unit. Retry is authorized whenever `surface.replaceGeneration` advances, including pruning-only progress on an otherwise indivisible surface; no replacement, recovery failure, an exhausted cap, cancellation, or an unknown/noncanonical error preserves the original provider failure. -- **Failure handling** — an unmatched `compact/start` is an inert crash marker because no replacement landed. Operational post-step failures warn and continue; overflow-recovery failure preserves the original provider error. +- **Overflow recovery** — below-threshold overflow bypasses normal retention and first prunes, then attempts one maximal balanced head reduction while leaving the newest indivisible unit. Retry is authorized whenever `surface.replaceGeneration` advances, including when pruning lands before later summary work throws. No replacement, an exhausted cap, cancellation, or an unknown/noncanonical error preserves the original provider failure. +- **Failure handling** — an unmatched `compact/start` is an inert crash marker because no summary replacement landed. Operational post-step failures warn and continue; overflow-recovery failure preserves the original provider error only when no earlier replacement advanced the surface. Cancellation remains authoritative after any progress. `summarize()` is the sole subclass hook. A template- or remote-summarizer subclass can override it while pressure, retention, provenance, shrink validation, and shadowed-token accounting stay on `ctx.tokenMeter`. The hook returns the summary blocks together with the call envelope it used (`{ summary, model, maxTokens? }`), which is logged on `compact/summary`. diff --git a/packages/compact/compact-basic/src/index.ts b/packages/compact/compact-basic/src/index.ts index 91e8ad0cea..34567f7e07 100644 --- a/packages/compact/compact-basic/src/index.ts +++ b/packages/compact/compact-basic/src/index.ts @@ -100,15 +100,28 @@ export class BasicCompactService extends CompactService { || retryAttempt >= this.config.maxOverflowRetries || signal.aborted) return next() - let generation: number + const generation = agent.session.surface.replaceGeneration let result: CompactionResult | null try { - generation = agent.session.surface.replaceGeneration result = await this.compactIfNeeded(agent, 'context-overflow', signal) } catch (recoveryError: unknown) { const message = recoveryError instanceof Error ? recoveryError.message : String(recoveryError) + // A model-free prune can land before later summary work fails. That + // durable reduction is sufficient retry proof; do not discard it just + // because the optional second phase threw. Cancellation still wins. + // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- signal can abort while recovery is awaited. + if (!signal.aborted && agent.session.surface.replaceGeneration > generation) { + ctx.logger.warn( + `context-overflow compaction failed after durable surface progress: ${message}; ` + + 'retrying from the replacement surface', + ) + return { action: 'retry' } + } ctx.logger.warn( - `context-overflow compaction failed: ${message}; preserving the original request error`, + // eslint-disable-next-line @typescript-eslint/no-unnecessary-condition -- signal can abort while recovery is awaited. + `context-overflow compaction failed: ${message}; ${signal.aborted + ? 'cancellation prevents retry' + : 'preserving the original request error'}`, ) return next() } diff --git a/packages/compact/compact-basic/tests/compact-basic.spec.ts b/packages/compact/compact-basic/tests/compact-basic.spec.ts index 190fb026e7..094a22a11e 100644 --- a/packages/compact/compact-basic/tests/compact-basic.spec.ts +++ b/packages/compact/compact-basic/tests/compact-basic.spec.ts @@ -1022,6 +1022,51 @@ describe('automatic listener and loader composition', () => { expect(compact.calls[0]!.text).toContain('tool result middle pruned') }) + it('retries from a durable prune when later overflow summarization throws', async () => { + const ctx = createContext(10_000) + const warnings: string[] = [] + ctx.logger.warn = ((message: string) => void warnings.push(message)) as typeof ctx.logger.warn + void new ToolResultPruneService(ctx, { + thresholdChars: 100, + headChars: 20, + tailChars: 10, + }) + const compact = new TestCompactService(ctx, { + thresholdRatio: 1, + retainTokens: 900, + }) + compact.error = new Error('summary unavailable after prune') + const session = oversizedToolResult(3_000, true) + + expect(await recover(ctx, agent(session, MODEL), overflow())).toEqual({ action: 'retry' }) + expect(session.surface.replaceGeneration).toBe(1) + expect(session.events.filter(event => event.type === 'tool/result')).toHaveLength(2) + expect(session.events.findLast(event => event.type === 'compact/end')?.data) + .toMatchObject({ error: 'summary unavailable after prune' }) + expect(warnings).toContainEqual(expect.stringContaining('retrying from the replacement surface')) + }) + + it('lets cancellation win when summary throws after a durable prune', async () => { + const ctx = createContext(10_000) + const controller = new AbortController() + void new ToolResultPruneService(ctx, { + thresholdChars: 100, + headChars: 20, + tailChars: 10, + }) + const compact = new TestCompactService(ctx, { + thresholdRatio: 1, + retainTokens: 900, + }) + compact.mutateDuringSummary = () => { controller.abort('cancelled during summary') } + compact.error = new Error('summary cancelled after prune') + const session = oversizedToolResult(3_000, true) + + expect(await recover(ctx, agent(session, MODEL), overflow(), 0, controller.signal)) + .toEqual({ action: 'fail' }) + expect(session.surface.replaceGeneration).toBe(1) + }) + it('preserves the newest whole tool-call/result pair during forced overflow compaction', async () => { const ctx = createContext() void new TestCompactService(ctx, { diff --git a/packages/support/invariants/README.md b/packages/support/invariants/README.md index 5ef2139db2..d80905e6ec 100644 --- a/packages/support/invariants/README.md +++ b/packages/support/invariants/README.md @@ -31,7 +31,7 @@ Session log (per session): - **turns pair and nest** — `turn/start` opens a turn, `turn/end` closes the matching one; no overlapping turns. - **steps nest in turns** — `step/start` opens a step in the open turn; `step/end` closes the matching step. - **chunks belong to an open step** — `step/start` precedes its `assistant/chunk`s. -- **an appended `tool/result` needs a prior `tool/call`** — fresh `surfaceOp: 'append'` results name the open step and consume its pending call, while a provenance-backed single-node `replace` is a turn-enclosed surface rewrite of an already-executed result. A `tool/call` may still have no result when the execution pipeline throws. +- **an appended `tool/result` needs a prior `tool/call`** — fresh `surfaceOp: 'append'` results name the open step and consume its pending call. A replacement exemption applies only to a provenance-backed rewrite of one current `tool/result` node whose complete data is identical except for `content`; it must still be turn-enclosed. A `tool/call` may still have no result when the execution pipeline throws. - **provenance sources are valid and unambiguous** — `sourceEventSeqs` contains unique earlier known seqs; only `assistant/message` may carry an explicit empty list, which denotes a known empty provider stream rather than absent legacy provenance. Agent status (per agent): diff --git a/packages/support/invariants/src/index.ts b/packages/support/invariants/src/index.ts index f33f7b4cec..0e12bfa246 100644 --- a/packages/support/invariants/src/index.ts +++ b/packages/support/invariants/src/index.ts @@ -7,6 +7,7 @@ * @module @deepseek-ai/dsh-invariants */ +import { isDeepStrictEqual } from 'node:util' import type { Context } from 'cordis' import { carrierKeyOf, isScopeCarrier } from '@deepseek-ai/dsh-scope' import { assertNever, HarnessError } from '@deepseek-ai/dsh-llm' @@ -50,13 +51,14 @@ interface SessionTrace { pendingCalls: Set /** Every seq seen so far — validates `sourceEventSeqs` references. */ knownSeqs: Set - /** - * The seqs currently on the surface linked list, in linked-list order - * (head to tail). A replace reorders this relative to seq order (the new - * node takes the replaced range's position), so range validation is - * positional, not by seq comparison. - */ - surface: number[] + /** Current surface nodes in linked-list order, with immutable event identity. */ + surface: SurfaceTraceNode[] +} + +/** Immutable identity retained only while an event is on the current surface. */ +interface SurfaceTraceNode { + seq: number + event: SessionEvent } /** One accepted event's deferred mutation of a live session trace. */ @@ -70,8 +72,9 @@ interface SessionTraceTransition { | { kind: 'clear' } /** The event's mutation of the derived surface order. */ surface: - | { kind: 'none' | 'append' } - | { kind: 'replace'; start: number; count: number } + | { kind: 'none' } + | { kind: 'append'; node: SurfaceTraceNode } + | { kind: 'replace'; start: number; count: number; node: SurfaceTraceNode } /** The committed event sequence to add to the known-sequence set. */ seq: number } @@ -85,6 +88,18 @@ function requireOpenStep(trace: SessionTrace, kind: string, turn: number, step: } } +/** Compare future-safe tool-result data while deliberately excluding content. */ +function sameToolResultDataExceptContent( + original: SessionEvent<'tool/result'>['data'], + replacement: SessionEvent<'tool/result'>['data'], +): boolean { + const originalRest = { ...original } as Record + const replacementRest = { ...replacement } as Record + delete originalRest['content'] + delete replacementRest['content'] + return isDeepStrictEqual(originalRest, replacementRest) +} + /** Validate one candidate event without mutating the committed session trace. */ function validateEvent(trace: SessionTrace, event: SessionEvent): SessionTraceTransition { // seq is strictly monotonic — the spine of replay equivalence. lastSeq @@ -139,14 +154,14 @@ function validateEvent(trace: SessionTrace, event: SessionEvent): SessionTraceTr // positional range — every shadowed node must appear in sourceEventSeqs. if (se.surfaceOp !== undefined) { if (se.surfaceOp === 'append') { - surface = { kind: 'append' } + surface = { kind: 'append', node: { seq: event.seq, event: se } } } else { const { start, end } = se.surfaceOp - const startIdx = trace.surface.indexOf(start) + const startIdx = trace.surface.findIndex(node => node.seq === start) if (startIdx === -1) { throw new InvariantError(`surface replace: start seq ${start} is not on the surface`) } - const endIdx = trace.surface.indexOf(end) + const endIdx = trace.surface.findIndex(node => node.seq === end) if (endIdx === -1) { throw new InvariantError(`surface replace: end seq ${end} is not on the surface`) } @@ -155,13 +170,18 @@ function validateEvent(trace: SessionTrace, event: SessionEvent): SessionTraceTr } // Every node the replace shadows (surface positions [startIdx, endIdx] // inclusive) must appear in sourceEventSeqs — the provenance contract. - const shadowed = trace.surface.slice(startIdx, endIdx + 1) + const shadowed = trace.surface.slice(startIdx, endIdx + 1).map(node => node.seq) const recorded = new Set(se.sourceEventSeqs ?? []) const missing = shadowed.filter(seq => !recorded.has(seq)) if (missing.length > 0) { throw new InvariantError(`surface replace: sourceEventSeqs must include every shadowed surface node; missing ${missing.join(', ')}`) } - surface = { kind: 'replace', start: startIdx, count: shadowed.length } + surface = { + kind: 'replace', + start: startIdx, + count: shadowed.length, + node: { seq: event.seq, event: se }, + } } } @@ -232,15 +252,26 @@ function validateEvent(trace: SessionTrace, event: SessionEvent): SessionTraceTr break } case 'tool/result': { - // A replacement rewrites an already-executed result whose recorded - // turn/step can be closed. Surface provenance above validates the rewrite; - // only fresh appends consume an open step's pending call. + // Only a content-only rewrite of one CURRENT tool-result node may bypass + // open-step/pending-call checks. The trace retains immutable surface event + // identity, so this validation never indexes a mutable or stale session. if (se.surfaceOp !== undefined && se.surfaceOp !== 'append') { if (trace.openTurn === null) { throw new InvariantError( 'tool/result surface replacement appended outside any open turn', ) } + const { start, end } = se.surfaceOp + if (start !== end) { + throw new InvariantError('tool/result surface replacement must rewrite exactly one current node') + } + const original = trace.surface.find(node => node.seq === start)?.event + if (original?.type !== 'tool/result') { + throw new InvariantError('tool/result surface replacement must target a current tool/result') + } + if (!sameToolResultDataExceptContent(original.data, event.data)) { + throw new InvariantError('tool/result surface replacement may change only content') + } break } requireOpenStep(trace, 'tool/result', event.data.turn, event.data.step) @@ -302,10 +333,14 @@ function applyTransition(trace: SessionTrace, transition: SessionTraceTransition case 'none': break case 'append': - trace.surface.push(transition.seq) + trace.surface.push(transition.surface.node) break case 'replace': - trace.surface.splice(transition.surface.start, transition.surface.count, transition.seq) + trace.surface.splice( + transition.surface.start, + transition.surface.count, + transition.surface.node, + ) break /* v8 ignore next -- validateEvent produces this closed transition union */ default: diff --git a/packages/support/invariants/tests/invariants.spec.ts b/packages/support/invariants/tests/invariants.spec.ts index 986cd93e7c..c55f3f1147 100644 --- a/packages/support/invariants/tests/invariants.spec.ts +++ b/packages/support/invariants/tests/invariants.spec.ts @@ -478,6 +478,39 @@ describe('HMR safety', () => { }) describe('surface invariants', () => { + async function toolResultRewriteFixture() { + const { ctx } = await setup() + const session = ctx.sessions.create() + session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) + const unrelated = session.append('user/message', { + content: [{ type: 'text', text: 'request' }], + source: { kind: 'user' }, + }, { surfaceOp: 'append' }) + session.append('step/start', { turn: 1, step: 1 }) + session.append('tool/call', { + turn: 1, + step: 1, + callId: CallId('rewrite'), + name: 'echo', + arguments: '{}', + }) + const originalData = { + turn: 1, + step: 1, + callId: CallId('rewrite'), + content: [{ type: 'text' as const, text: 'original' }], + isError: true, + error: { name: 'ExitError', code: 'EXIT_1' }, + meta: { presentation: { kind: 'terminal', output: 'full output' } }, + futureField: { nested: ['preserve', 1] }, + } + const original = session.append('tool/result', originalData, { surfaceOp: 'append' }) + session.append('step/end', { turn: 1, step: 1 }) + session.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) + session.append('turn/start', { turn: 2, trigger: { kind: 'message', source: { kind: 'user' } } }) + return { session, unrelated, original } + } + it('accepts well-formed surface metadata', async () => { const { ctx } = await setup() const session = ctx.sessions.create() @@ -501,27 +534,7 @@ describe('surface invariants', () => { }) it('treats a provenance-backed tool-result replacement as a turn-enclosed rewrite', async () => { - const { ctx } = await setup() - const session = ctx.sessions.create() - session.append('turn/start', { turn: 1, trigger: { kind: 'message', source: { kind: 'user' } } }) - session.append('step/start', { turn: 1, step: 1 }) - session.append('tool/call', { - turn: 1, - step: 1, - callId: CallId('rewrite'), - name: 'echo', - arguments: '{}', - }) - const original = session.append('tool/result', { - turn: 1, - step: 1, - callId: CallId('rewrite'), - content: [{ type: 'text', text: 'original' }], - isError: false, - }, { surfaceOp: 'append' }) - session.append('step/end', { turn: 1, step: 1 }) - session.append('turn/end', { turn: 1, reason: { kind: 'completed' } }) - session.append('turn/start', { turn: 2, trigger: { kind: 'message', source: { kind: 'user' } } }) + const { session, original } = await toolResultRewriteFixture() expect(() => session.append('tool/result', { ...original.data, @@ -532,6 +545,47 @@ describe('surface invariants', () => { })).not.toThrow() }) + it('rejects a tool-result replacement targeting an unrelated current node', async () => { + const { session, unrelated, original } = await toolResultRewriteFixture() + expect(() => session.append('tool/result', { + ...original.data, + content: [{ type: 'text', text: 'forged' }], + }, { + surfaceOp: { op: 'replace', start: unrelated.seq, end: unrelated.seq }, + sourceEventSeqs: [unrelated.seq], + })).toThrow(/must target a current tool\/result/) + }) + + it('rejects a multi-node tool-result replacement even with complete provenance', async () => { + const { session, unrelated, original } = await toolResultRewriteFixture() + expect(() => session.append('tool/result', { + ...original.data, + content: [{ type: 'text', text: 'forged' }], + }, { + surfaceOp: { op: 'replace', start: unrelated.seq, end: original.seq }, + sourceEventSeqs: [unrelated.seq, original.seq], + })).toThrow(/must rewrite exactly one current node/) + }) + + it.each([ + ['callId', { callId: CallId('forged') }], + ['turn', { turn: 2 }], + ['step', { step: 2 }], + ['error', { error: { name: 'ExitError', code: 'DIFFERENT' } }], + ['meta', { meta: { presentation: { kind: 'generic' } } }], + ['future data', { futureField: { nested: ['changed'] } }], + ])('rejects a content rewrite with altered %s', async (_label, altered) => { + const { session, original } = await toolResultRewriteFixture() + expect(() => session.append('tool/result', { + ...original.data, + ...altered, + content: [{ type: 'text', text: 'pruned' }], + }, { + surfaceOp: { op: 'replace', start: original.seq, end: original.seq }, + sourceEventSeqs: [original.seq], + })).toThrow(/may change only content/) + }) + it('accepts known-empty assistant provenance and rejects empty provenance elsewhere', async () => { const { ctx } = await setup() const session = ctx.sessions.create() diff --git a/packages/ui/acp/README.md b/packages/ui/acp/README.md index 3650cf27d4..15733cb9d9 100644 --- a/packages/ui/acp/README.md +++ b/packages/ui/acp/README.md @@ -99,7 +99,7 @@ The JSON-RPC frames go on stdout, so this plugin MUST run in an example that loa **What the model sees**: When optional consumers are loaded, ACP form answers become the exact JSON shape documented by `dsh-tool-ask-user`. Failures become `Error: ACP user questions must come from an agent-owned request`, `Error: ACP user question has no matching session`, `Error: ACP elicitation request failed`, `Error: ask_user_question was cancelled by the user`, `Error: ask_user_question returned no answer`, or `Error: ask_user_question was aborted before the user answered`. Permission decisions control whether another tool yields success or denial. ACP tool cards, terminal output, diffs, and streamed session updates are UI-only. -**Token effect**: Answer, error, and denial text enters context only through the owning tool result; presentation metadata adds zero model tokens. +**Token effect**: Answer, error, and denial text enters context only through the owning tool result; presentation metadata adds zero model tokens. A replacement `tool/result` still changes the model-facing session surface, but live and replayed ACP feeds ignore it as an execution update so the original terminal or diff completion is not overwritten. ### Permission preset switches diff --git a/packages/ui/acp/acp-feature-support.md b/packages/ui/acp/acp-feature-support.md index 29b0abc073..f1c3a918d5 100644 --- a/packages/ui/acp/acp-feature-support.md +++ b/packages/ui/acp/acp-feature-support.md @@ -82,7 +82,7 @@ These are capabilities the bridge would *drive* on the editor. The harness runs | `agent_thought_chunk` | S | ✅ | ✅ | ✅ | From `assistant/chunk` reasoning-delta. | | `user_message_chunk` | S | ✅ | ✅ | ✅ | Emitted during `session/load` replay to reconstruct the user side. | | `tool_call` | S | ✅ | ✅ | ✅ | Tool-owned presentation (`presentCall`); see [§5](#5-tool-call-rendering). | -| `tool_call_update` | S | ✅ | ✅ | ✅ | From `tool/result` via `presentResult`. | +| `tool_call_update` | S | ✅ | ✅ | ✅ | From appended `tool/result` via `presentResult`; replacement results rewrite model context and do not duplicate or overwrite execution presentation. | | `plan` | S | ❌ | ✅ | ✅ | No agent plan emitted. Both adapters emit real plan entries (Codex's `CodexEventHandler.updatePlan` maps `turn/plan/updated` → `{ sessionUpdate: 'plan', entries }`). | | `available_commands_update` | S | ❌ | ✅ | ✅ | No slash commands advertised. | | `current_mode_update` | S | ❌ | ✅ | ✅ | No session modes. | diff --git a/packages/ui/acp/src/index.ts b/packages/ui/acp/src/index.ts index c464ffd22b..66719d0ddd 100644 --- a/packages/ui/acp/src/index.ts +++ b/packages/ui/acp/src/index.ts @@ -904,7 +904,8 @@ function validateMcpServers(params: { mcpServers?: unknown[] }): void { * loaded transcript reconstructs the USER side of each turn without echoing * a live `session/prompt` back to the client * - `tool/call` → `tool_call` (pending) - * - `tool/result` → `tool_call_update` (completed/failed) + * - appended `tool/result` → `tool_call_update` (completed/failed) + * - replacement `tool/result` → no update (context rewrite, not execution) * * Tool-call presentation (title/kind/rawInput, and the completed-state content) * is owned by each TOOL via `presentCall`/`presentResult` — the bridge never @@ -965,6 +966,10 @@ export function streamSessionEventUpdate( return } case 'tool/result': { + // Replacements (for example model-free pruning) are transcript rewrites, + // not repeated tool executions. Re-presenting one would consume no + // pending call and could clobber the original terminal/diff completion. + if (event.surfaceOp !== undefined && event.surfaceOp !== 'append') return const view = presenter.result(event.data.callId, event.data.content, event.data.isError, event.data.meta) notify({ sessionId, update: toolResultUpdate(event.data.callId, view, event.data.isError, terminal) }) return diff --git a/packages/ui/acp/tests/load.spec.ts b/packages/ui/acp/tests/load.spec.ts index f57767fb38..4c22681b79 100644 --- a/packages/ui/acp/tests/load.spec.ts +++ b/packages/ui/acp/tests/load.spec.ts @@ -162,6 +162,53 @@ describe('acp bridge — session/load replay', () => { expect(meta.terminal_exit?.exit_code).toBe(0) }) + it('keeps one terminal completion live and on replay when a pruning replacement is logged', async () => { + live = await makeBridgeHarness({ + storageDir, + withBash: true, + script: [toolCallResponse('c1', 'bash', { command: 'echo full', description: 'Print full output' }), textResponse('done')], + }) + await live.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: { _meta: { terminal_output: true } } }) + const { sessionId } = await live.client.newSession({ cwd: process.cwd(), mcpServers: [] }) + await live.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'run it' }] }) + + const session = live.ctx.agents.get(AgentId(sessionId))!.session + const original = session.events.find(event => event.type === 'tool/result') + if (original?.type !== 'tool/result') throw new Error('expected original tool/result') + const liveCompletions = () => live!.updates.filter(update => + update.sessionUpdate === 'tool_call_update' && update.toolCallId === 'c1') + expect(liveCompletions()).toHaveLength(1) + expect((liveCompletions()[0] as { _meta?: { terminal_output?: { data: string } } })._meta?.terminal_output?.data) + .toBe('full\n') + + session.append('turn/start', { turn: 2, trigger: { kind: 'message', source: { kind: 'user' } } }) + session.append('tool/result', { + ...original.data, + content: [{ type: 'text', text: '[... tool result middle pruned ...]' }], + }, { + surfaceOp: { op: 'replace', start: original.seq, end: original.seq }, + sourceEventSeqs: [original.seq], + }) + session.append('turn/end', { turn: 2, reason: { kind: 'completed' } }) + + // The replacement is durable but is not another live completion. + expect(session.events.filter(event => event.type === 'tool/result')).toHaveLength(2) + expect(JSON.stringify(session.deriveMessages())).toContain('tool result middle pruned') + expect(liveCompletions()).toHaveLength(1) + await live.dispose() + live = undefined + + loader = await makeBridgeHarness({ storageDir, withBash: true, script: [] }) + await loader.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: { _meta: { terminal_output: true } } }) + await loader.client.loadSession({ sessionId, cwd: process.cwd(), mcpServers: [] }) + + const replayed = loader.updates.filter(update => + update.sessionUpdate === 'tool_call_update' && update.toolCallId === 'c1') + expect(replayed).toHaveLength(1) + expect((replayed[0] as { _meta?: { terminal_output?: { data: string } } })._meta?.terminal_output?.data) + .toBe('full\n') + }) + it('a load whose resume finishes after a client disconnect leaks no live session', async () => { // Stall persistence so transport closes while resume is pending. Whether the SDK rejects first // or the bridge's post-await guard fires, no agent may survive for the dead connection. diff --git a/packages/ui/acp/tests/stream-update.spec.ts b/packages/ui/acp/tests/stream-update.spec.ts index 2afa49e1d4..c2884e0b1f 100644 --- a/packages/ui/acp/tests/stream-update.spec.ts +++ b/packages/ui/acp/tests/stream-update.spec.ts @@ -105,6 +105,22 @@ describe('streamSessionEventUpdate', () => { expect((failed[0] as { status: string }).status).toBe('failed') }) + it('emits no execution update for a tool-result surface replacement', () => { + const replacement = { + ...evt('tool/result', { + turn: 1, + step: 1, + callId: CallId('c1'), + content: [{ type: 'text', text: '[... tool result middle pruned ...]' }], + isError: false, + }), + seq: 2, + surfaceOp: { op: 'replace', start: 1, end: 1 }, + sourceEventSeqs: [1], + } as SessionEvent + expect(updatesFor(replacement)).toEqual([]) + }) + it('drops non-text tool-result content (text-only)', () => { const update = updatesFor(evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), @@ -450,6 +466,16 @@ describe('terminal-card mapping (capability-gated)', () => { const callEvent = evt('tool/call', { turn: 1, step: 1, callId: CallId('c1'), name: 'bash', arguments: JSON.stringify({ command: 'echo hi', description: 'Greet' }) }) const resultEvent = evt('tool/result', { turn: 1, step: 1, callId: CallId('c1'), content: [{ type: 'text', text: 'hi\n' }], isError: false }) + const prunedResultEvent = { + ...resultEvent, + seq: 2, + data: { + ...resultEvent.data, + content: [{ type: 'text', text: '[... tool result middle pruned ...]' }], + }, + surfaceOp: { op: 'replace', start: 1, end: 1 }, + sourceEventSeqs: [1], + } as SessionEvent function termUpdates(tool: ToolDefinition, enabled: boolean, cwd: string | undefined, ...events: SessionEvent[]): SessionNotification['update'][] { const presenter = new ToolPresenter(registryOf(tool)) @@ -477,6 +503,27 @@ describe('terminal-card mapping (capability-gated)', () => { }) }) + it('live/replay translation preserves the original terminal completion across a pruning rewrite', () => { + const updates = termUpdates( + termTool({ card: 'terminal' }, { output: 'hi\n', exitCode: 0 }), + true, + '/work/proj', + callEvent, + resultEvent, + prunedResultEvent, + ) + expect(updates).toHaveLength(2) + expect(updates[1]).toEqual({ + sessionUpdate: 'tool_call_update', + toolCallId: 'c1', + status: 'completed', + _meta: { + terminal_output: { terminal_id: 'c1', data: 'hi\n' }, + terminal_exit: { terminal_id: 'c1', exit_code: 0 }, + }, + }) + }) + it('capability ON: an ABSOLUTE tool cwd wins; a RELATIVE one resolves against the session cwd', () => { const [absCall] = termUpdates(termTool({ card: 'terminal', cwd: '/explicit/abs' }, { output: 'x' }), true, '/work/proj', callEvent) expect((absCall as unknown as { _meta: { terminal_info: { cwd: string } } })._meta.terminal_info.cwd).toBe('/explicit/abs') @@ -633,17 +680,38 @@ describe('result-time diff card (REAL fs edit tool → tool_call_update diff blo // call-time snippet, then the tool/result carries the tool's computed applied-hunk `meta`, // which presentResult narrows into a `diff` result card the bridge forwards as `{ type: // 'diff' }` content blocks. The real tool is required because its result metadata is the contract. - it('forwards the applied-hunk meta onto the wire as tool_call_update diff content', async () => { + it('live/replay translation keeps the applied diff when a pruning rewrite follows', async () => { const ctx = await fsCtx() const presenter = new ToolPresenter(ctx.tools) const args = JSON.stringify({ file_path: 'src/b.ts', old_string: 'OLD', new_string: 'NEW' }) // The applied hunk the tool would compute and persist on the result meta. const meta = { diffs: [{ path: 'src/b.ts', oldText: 'a\nOLD\nb', newText: 'a\nNEW\nb' }] } - const [, resultUpdate] = updatesWith( + const originalResult = evt('tool/result', { + turn: 1, + step: 1, + callId: CallId('e1'), + content: [{ type: 'text', text: 'ok' }], + isError: false, + meta, + }) + const replacement = { + ...originalResult, + seq: 3, + data: { + ...originalResult.data, + content: [{ type: 'text', text: '[... tool result middle pruned ...]' }], + }, + surfaceOp: { op: 'replace', start: 2, end: 2 }, + sourceEventSeqs: [2], + } as SessionEvent + const updates = updatesWith( presenter, evt('tool/call', { turn: 1, step: 1, callId: CallId('e1'), name: 'edit', arguments: args }), - evt('tool/result', { turn: 1, step: 1, callId: CallId('e1'), content: [{ type: 'text', text: 'ok' }], isError: false, meta }), + originalResult, + replacement, ) + expect(updates).toHaveLength(2) + const resultUpdate = updates[1] expect(resultUpdate).toEqual({ sessionUpdate: 'tool_call_update', toolCallId: 'e1', diff --git a/packages/ui/stdio/README.md b/packages/ui/stdio/README.md index b7d320880d..2327857b82 100644 --- a/packages/ui/stdio/README.md +++ b/packages/ui/stdio/README.md @@ -27,7 +27,7 @@ The plugin seeds display labels from the live agent registry, then tracks `agent **What the model sees**: Each non-empty terminal line outside an active question becomes one text block, sent with `agent.send()` while the target agent is idle and `agent.steer()` while it is running. -**Token effect**: Submitted text is retained under the agent loop's normal session-history and compaction rules. The welcome banner, `> ` prompt, rendered transcript, and `[tool call]` / `[tool result]` terminal lines add no tokens. +**Token effect**: Submitted text is retained under the agent loop's normal session-history and compaction rules. The welcome banner, `> ` prompt, rendered transcript, and `[tool call]` / `[tool result]` terminal lines add no tokens. A replacement `tool/result` remains model-visible through the session surface but is not rendered as a second execution; stdio keeps the original full-fidelity result line. ### Terminal user-interaction answers diff --git a/packages/ui/stdio/src/index.ts b/packages/ui/stdio/src/index.ts index 1e665381ce..aea10827c7 100644 --- a/packages/ui/stdio/src/index.ts +++ b/packages/ui/stdio/src/index.ts @@ -124,6 +124,10 @@ export function createStdioChat(ctx: Context, config: Config, runtime: StdioRunt inReasoning = false output.write(`\n [tool call] ${toolName}(${args})`) } else if (event.type === 'tool/result') { + // A surface replacement changes future model context; it is not another + // execution. Keep the original full-fidelity terminal presentation and + // suppress duplicate output during live delivery or log replay. + if (event.surfaceOp !== undefined && event.surfaceOp !== 'append') return const { content } = event.data const text = content.filter(block => block.type === 'text').map(block => block.text).join('') output.write(`\n [tool result] ${text}\n `) diff --git a/packages/ui/stdio/tests/stdio.spec.ts b/packages/ui/stdio/tests/stdio.spec.ts index 7bb6a6f245..f757f62401 100644 --- a/packages/ui/stdio/tests/stdio.spec.ts +++ b/packages/ui/stdio/tests/stdio.spec.ts @@ -288,6 +288,43 @@ describe('createStdioChat rendering', () => { expect(out.text()).toContain('[tool result] file.txt') }) + it('renders one full-fidelity result whether the event feed is live or replayed', async () => { + const { ctx, out } = await setup() + const session = makeSession('main') + const original = { + type: 'tool/result', + seq: 2, + time: 0, + data: { + turn: 1, + step: 1, + callId: 'c1', + content: [{ type: 'text', text: 'full terminal output' }], + isError: false, + meta: { terminal: { output: 'full terminal output' } }, + }, + surfaceOp: 'append', + } as SessionEvent + const replacement = { + ...original, + seq: 3, + data: { + ...original.data, + content: [{ type: 'text', text: '[... tool result middle pruned ...]' }], + }, + surfaceOp: { op: 'replace', start: 2, end: 2 }, + sourceEventSeqs: [2], + } as SessionEvent + + // Stdio consumes the same session/event shape whether a host forwards a + // live append or replays a stored log through the rendering feed. + for (const event of [original, replacement]) ctx.emit('session/event', session, event) + + expect(out.text().match(/\[tool result\]/g)).toHaveLength(1) + expect(out.text()).toContain('full terminal output') + expect(out.text()).not.toContain('tool result middle pruned') + }) + it('renders a todo/write session event as a glyphed checklist', async () => { const { ctx, out } = await setup() const session = {} as Session