From 3e1c63b2eaa68f170a6ce7bfa25515f7eb8d6f4a Mon Sep 17 00:00:00 2001 From: Yichen Jiang Date: Thu, 6 Aug 2026 15:47:42 +0800 Subject: [PATCH] test(web): stop the stats line's wall-clock segments from deciding a golden MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `StatsLine` renders its LLM, tool-call, and throughput segments only while the matching measurement exceeds zero, and all three are wall clock taken during the replay. A machine that finishes a step inside one millisecond drops the segment a slower one keeps, so a golden recorded what the recording machine's speed was rather than what the page shows. Goldens across this suite already disagreed about the LLM segment for that reason, and CI failed on whichever test landed on a slow enough runner — a different test each run, always the same one-line difference. Tokenizing the values was never enough, because presence is what moves. The normalizer now drops those segments outright, each taking one adjacent separator so nothing is left holding a dangling separator or a doubled space, and the recorded goldens are normalized the same way. `TTFT avg` stays: it gates on a step count the fixture determines. --- apps/web/tests/scaffold.ts | 20 ++++++++++++++++++- .../snapshots/bash-abort-row/ui.expected.md | 2 +- .../snapshots/code-mode-round/ui.expected.md | 4 ++-- .../cordis-tool-round/ui.expected.md | 4 ++-- .../snapshots/fresh-round-trip/ui.expected.md | 4 ++-- .../lifecycle-chrome/reloaded.expected.md | 4 ++-- .../live-interactions/retry.expected.md | 4 ++-- .../snapshots/math-rendering/ui.expected.md | 2 +- .../snapshots/message-actions/ui.expected.md | 4 ++-- .../plan-review/approved.expected.md | 4 ++-- .../question-composer/answered.expected.md | 4 ++-- .../seeded-history/command-row.expected.md | 4 ++-- .../snapshots/seeded-history/ui.expected.md | 4 ++-- .../snapshots/steering/settled.expected.md | 4 ++-- .../subagent-conversation/ui.expected.md | 6 +++--- .../snapshots/web-search-round/ui.expected.md | 4 ++-- 16 files changed, 48 insertions(+), 30 deletions(-) diff --git a/apps/web/tests/scaffold.ts b/apps/web/tests/scaffold.ts index 52eb7f151d..0488b9cecf 100644 --- a/apps/web/tests/scaffold.ts +++ b/apps/web/tests/scaffold.ts @@ -533,13 +533,24 @@ export async function seedSession(scaffold: WebScaffold, fixtureText: string, id /** * Normalize an aria snapshot: uuid, cwd, workspace-basename, duration, and - * decode-throughput volatility collapse to stable tokens. + * decode-throughput volatility collapse to stable tokens, and the stats line's + * wall-clock-gated segments drop out entirely. * * Throughput needs a token for the same reason durations do, and no fixture * can supply one: the figure divides a replayed step's output tokens by the * wall time the local run took to stream them, so it moves between two runs * on one machine (measured 69 → 70 tok/s) and swings wildly on a fast replay * (26333 tok/s for a 3 ms stream). + * + * Tokenizing those values is not enough, because `StatsLine` renders each such + * segment only while its measurement exceeds zero (`llmMs`, `toolMs`, and + * `decodeMs` all gate on `> 0`). A replay that finishes a step inside one + * millisecond therefore omits the segment a slower machine keeps, and the + * golden would record how fast the recording machine was rather than what the + * page shows: goldens recorded across this suite disagree on the `LLM` segment + * for exactly that reason, and CI failed on whichever test happened to run on + * a slow enough runner. Dropping the segments makes presence stop deciding. + * `TTFT avg` stays: it gates on a step count the fixture determines. */ function normalizeAria(snapshot: string, workspaceCwd: string): string { // The session heading renders the workspace's basename, not the full @@ -560,6 +571,13 @@ function normalizeAria(snapshot: string, workspaceCwd: string): string { duration => duration.startsWith('约') ? duration : '{{duration}}', ) .replace(/\d+(?:\.\d+)?(?= tok\/s(?!\w))/g, '{{throughput}}') + // Each removal takes one adjacent separator with it, so nothing is left + // holding a dangling `·` or a doubled space: a segment followed by its + // intra-group separator loses that, and one ending its group loses the + // space in front of it instead. + .replace(/(?:LLM|Tool call|工具调用) \{\{duration\}\} · /g, '') + .replace(/ (?:LLM|Tool call|工具调用) \{\{duration\}\}/g, '') + .replace(/ (?:· )?\{\{throughput\}\} tok\/s/g, '') // Message IconActions clocks widen by calendar day/year; collapse every // shape so goldens stay stable across midnight and year boundaries. .replace(/\d{4}年\d{1,2}月\d{1,2}日 \d{2}:\d{2}/g, '{{clock}}') diff --git a/apps/web/tests/snapshots/bash-abort-row/ui.expected.md b/apps/web/tests/snapshots/bash-abort-row/ui.expected.md index 1b9e6aa339..3066f0ffc0 100644 --- a/apps/web/tests/snapshots/bash-abort-row/ui.expected.md +++ b/apps/web/tests/snapshots/bash-abort-row/ui.expected.md @@ -30,4 +30,4 @@ - text: Select model - img - button "Send message" [disabled] -- text: 1 turns · 1 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 0% Input 10 tok · Output 10 tok +- text: 1 turns · 1 steps TTFT avg {{duration}} Cache hit 0% Input 10 tok · Output 10 tok diff --git a/apps/web/tests/snapshots/code-mode-round/ui.expected.md b/apps/web/tests/snapshots/code-mode-round/ui.expected.md index 0c2cf8604c..cb5bd25a4e 100644 --- a/apps/web/tests/snapshots/code-mode-round/ui.expected.md +++ b/apps/web/tests/snapshots/code-mode-round/ui.expected.md @@ -36,7 +36,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -46,4 +46,4 @@ - img - button "7% of context used" - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 52% Input 17.2K tok · Output 252 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 52% Input 17.2K tok · Output 252 tok diff --git a/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md b/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md index 33b1d6cd0f..44255e98ec 100644 --- a/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md +++ b/apps/web/tests/snapshots/cordis-tool-round/ui.expected.md @@ -51,7 +51,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -61,4 +61,4 @@ - img - button "13% of context used" - button "Send message" [disabled] -- text: 1 turns · 4 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 77% Input 66.5K tok · Output 312 tok +- text: 1 turns · 4 steps TTFT avg {{duration}} Cache hit 77% Input 66.5K tok · Output 312 tok diff --git a/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md b/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md index aebc2a45b6..d0d9ee2632 100644 --- a/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md +++ b/apps/web/tests/snapshots/fresh-round-trip/ui.expected.md @@ -31,7 +31,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -41,4 +41,4 @@ - img - button "6% of context used" - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 99% Input 15.7K tok · Output 111 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 99% Input 15.7K tok · Output 111 tok diff --git a/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md b/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md index 6b6671ec01..8b82c0b746 100644 --- a/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md +++ b/apps/web/tests/snapshots/lifecycle-chrome/reloaded.expected.md @@ -23,7 +23,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -33,4 +33,4 @@ - img - button "6% of context used" - button "Send message" [disabled] -- text: 1 turns · 1 steps LLM {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 99% Input 7.8K tok · Output 21 tok +- text: 1 turns · 1 steps TTFT avg {{duration}} Cache hit 99% Input 7.8K tok · Output 21 tok diff --git a/apps/web/tests/snapshots/live-interactions/retry.expected.md b/apps/web/tests/snapshots/live-interactions/retry.expected.md index f127d3e8d1..1442b28bb9 100644 --- a/apps/web/tests/snapshots/live-interactions/retry.expected.md +++ b/apps/web/tests/snapshots/live-interactions/retry.expected.md @@ -25,7 +25,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -35,4 +35,4 @@ - img - button "6% of context used" - button "Send message" [disabled] -- text: 1 turns · 1 steps LLM {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 99% Input 7.8K tok · Output 79 tok +- text: 1 turns · 1 steps TTFT avg {{duration}} Cache hit 99% Input 7.8K tok · Output 79 tok diff --git a/apps/web/tests/snapshots/math-rendering/ui.expected.md b/apps/web/tests/snapshots/math-rendering/ui.expected.md index be1bbb7069..69af6a87ac 100644 --- a/apps/web/tests/snapshots/math-rendering/ui.expected.md +++ b/apps/web/tests/snapshots/math-rendering/ui.expected.md @@ -44,4 +44,4 @@ - text: Select model - img - button "Send message" [disabled] -- text: 1 turns · 1 steps LLM {{duration}} Input 0 tok · Output 0 tok +- text: 1 turns · 1 steps Input 0 tok · Output 0 tok diff --git a/apps/web/tests/snapshots/message-actions/ui.expected.md b/apps/web/tests/snapshots/message-actions/ui.expected.md index 81c2796e5a..1e27bbc664 100644 --- a/apps/web/tests/snapshots/message-actions/ui.expected.md +++ b/apps/web/tests/snapshots/message-actions/ui.expected.md @@ -20,7 +20,7 @@ - img - button "Branch into a new conversation" [disabled]: - img -- text: Available only on the last message of a completed turn 7/25 {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: Available only on the last message of a completed turn 7/25 {{clock}} Ran for {{duration}} TTFT {{duration}} - button "Read a.txt": - img - img @@ -55,4 +55,4 @@ - text: Select model - img - button "Send message" [disabled] -- text: 2 turns · 3 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 98% Input 7.8K tok · Output 103 tok +- text: 2 turns · 3 steps TTFT avg {{duration}} Cache hit 98% Input 7.8K tok · Output 103 tok diff --git a/apps/web/tests/snapshots/plan-review/approved.expected.md b/apps/web/tests/snapshots/plan-review/approved.expected.md index f0c7d718e0..e393ae6c7b 100644 --- a/apps/web/tests/snapshots/plan-review/approved.expected.md +++ b/apps/web/tests/snapshots/plan-review/approved.expected.md @@ -36,7 +36,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -46,4 +46,4 @@ - img - button "4% of context used" - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 51% Input 10.2K tok · Output 346 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 51% Input 10.2K tok · Output 346 tok diff --git a/apps/web/tests/snapshots/question-composer/answered.expected.md b/apps/web/tests/snapshots/question-composer/answered.expected.md index 82e0b468c1..df985ee3ff 100644 --- a/apps/web/tests/snapshots/question-composer/answered.expected.md +++ b/apps/web/tests/snapshots/question-composer/answered.expected.md @@ -31,7 +31,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -41,4 +41,4 @@ - img - button "3% of context used" - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 95% Input 8.6K tok · Output 180 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 95% Input 8.6K tok · Output 180 tok diff --git a/apps/web/tests/snapshots/seeded-history/command-row.expected.md b/apps/web/tests/snapshots/seeded-history/command-row.expected.md index 467a4364b8..8faaa80c9f 100644 --- a/apps/web/tests/snapshots/seeded-history/command-row.expected.md +++ b/apps/web/tests/snapshots/seeded-history/command-row.expected.md @@ -33,7 +33,7 @@ - img - button "Branch into a new conversation": - img -- text: 7/25 {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: 7/25 {{clock}} Ran for {{duration}} TTFT {{duration}} - button "Context compacted View compaction summary": - img - text: Context compacted View compaction summary @@ -51,4 +51,4 @@ - text: Select model - img - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 98% Input 15.8K tok · Output 135 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 98% Input 15.8K tok · Output 135 tok diff --git a/apps/web/tests/snapshots/seeded-history/ui.expected.md b/apps/web/tests/snapshots/seeded-history/ui.expected.md index 55fcb89ec8..1eb883fdb4 100644 --- a/apps/web/tests/snapshots/seeded-history/ui.expected.md +++ b/apps/web/tests/snapshots/seeded-history/ui.expected.md @@ -33,7 +33,7 @@ - img - button "Branch into a new conversation": - img -- text: 7/25 {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: 7/25 {{clock}} Ran for {{duration}} TTFT {{duration}} - button "Context compacted View compaction summary": - img - text: Context compacted View compaction summary @@ -49,4 +49,4 @@ - text: Select model - img - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 98% Input 15.8K tok · Output 135 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 98% Input 15.8K tok · Output 135 tok diff --git a/apps/web/tests/snapshots/steering/settled.expected.md b/apps/web/tests/snapshots/steering/settled.expected.md index 77385c6333..243ca85ffc 100644 --- a/apps/web/tests/snapshots/steering/settled.expected.md +++ b/apps/web/tests/snapshots/steering/settled.expected.md @@ -37,7 +37,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -47,4 +47,4 @@ - img - button "6% of context used" - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 98% Input 15.8K tok · Output 156 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 98% Input 15.8K tok · Output 156 tok diff --git a/apps/web/tests/snapshots/subagent-conversation/ui.expected.md b/apps/web/tests/snapshots/subagent-conversation/ui.expected.md index a01eea56d8..d3be54aa1d 100644 --- a/apps/web/tests/snapshots/subagent-conversation/ui.expected.md +++ b/apps/web/tests/snapshots/subagent-conversation/ui.expected.md @@ -28,7 +28,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s Now give the same explanation to a human reader. {{clock}} +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} Now give the same explanation to a human reader. {{clock}} - button "Copy": - img - button "Branch into a new conversation" [disabled]: @@ -43,11 +43,11 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img - 'button "Access mode, current: Workspace Write"': Workspace Write - button "6% of context used" - button "Send message" [disabled] -- text: 2 turns · 2 steps LLM {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 99% Input 15.6K tok · Output 158 tok +- text: 2 turns · 2 steps TTFT avg {{duration}} Cache hit 99% Input 15.6K tok · Output 158 tok diff --git a/apps/web/tests/snapshots/web-search-round/ui.expected.md b/apps/web/tests/snapshots/web-search-round/ui.expected.md index 1e2dcf9eca..9995c3050f 100644 --- a/apps/web/tests/snapshots/web-search-round/ui.expected.md +++ b/apps/web/tests/snapshots/web-search-round/ui.expected.md @@ -23,7 +23,7 @@ - img - button "Branch into a new conversation": - img -- text: {{clock}} Ran for {{duration}} TTFT {{duration}} {{throughput}} tok/s +- text: {{clock}} Ran for {{duration}} TTFT {{duration}} - textbox "Message the agent" - button "Commands": - img @@ -33,4 +33,4 @@ - img - button "0% of context used" - button "Send message" [disabled] -- text: 1 turns · 2 steps LLM {{duration}} · Tool call {{duration}} TTFT avg {{duration}} · {{throughput}} tok/s Cache hit 0% Input 22 tok · Output 7 tok +- text: 1 turns · 2 steps TTFT avg {{duration}} Cache hit 0% Input 22 tok · Output 7 tok