mirror of
https://github.com/deepseek-ai/deepseek-harness
synced 2026-08-15 21:04:50 +00:00
Reform the compaction blueprint so a runaway turn survives and the design
stops drifting across review rounds:
- Drop in-flight-turn protection ("layer 2"). Retention is a uniform tail→head
whole-unit walk; the only structural guard is step-alignment. A single turn
that alone exceeds the window now compacts its own early closed steps instead
of being retained verbatim (the failure mode that motivated this).
- Move auto-compaction off the agent/request waterfall onto a new awaited
agent/pre-request loop seam, fired before history derivation. Compaction
mutates the surface; the loop derives once from the result — no double-derive,
and a listener structurally cannot act on not-yet-derived messages.
- Tighten compactIfNeeded to required (session, system, model, signal).
- Enforce a single-pass convergence invariant in resolveConfig: reject configs
where summarizationMaxTokens + retainTokens exceeds the threshold, so a
compaction can never immediately re-trigger.
- Document the crash vs recoverable failure taxonomy; core session repair stays
compaction-agnostic (a log-only orphaned compact/start is inert).
- Wire dsh-compact-basic into examples/coding-agent and add a with-key
compaction e2e (compaction's first real-world exercise + runaway net).
- Rewrite the RFC to encode the blueprint and move it to implemented/.
The runaway-turn snapshot is a named deferred follow-up: dsh-llm-replay cannot
yet serve the interleaved summarization model call.
72 lines
3.2 KiB
TypeScript
72 lines
3.2 KiB
TypeScript
import { mkdtemp, rm } from 'node:fs/promises'
|
|
import { tmpdir } from 'node:os'
|
|
import { join } from 'node:path'
|
|
import { afterEach, describe, expect, it } from 'vitest'
|
|
import type { Context } from 'cordis'
|
|
import type { ReactLoopAgent } from '@deepseek-ai/dsh-agent-loop'
|
|
import { AgentId } from '@deepseek-ai/dsh-agent'
|
|
import { SessionId } from '@deepseek-ai/dsh-session'
|
|
import { codingHarness, finalText, SYSTEM_PROMPT, waitForIdle } from './harness.ts'
|
|
|
|
/**
|
|
* Proves durable conversation continuity end-to-end: run 1 tells the REAL model
|
|
* a fact and persists the turn to JSONL; run 2 is a fresh harness (new Context,
|
|
* same `.sessions` root) that RESUMES the persisted session id and asks the
|
|
* model to recall the fact. The recall can only come from the rehydrated event
|
|
* log — a fresh session would have no idea. Key-gated like the other e2es.
|
|
*/
|
|
|
|
const SECRET = 'plum-galaxy-1791'
|
|
const SESSION_ID = SessionId('resume-e2e-session')
|
|
|
|
let ctx: Context | undefined
|
|
let root: string | undefined
|
|
|
|
afterEach(async () => {
|
|
// Dispose even on failure/retry: agent-loop teardown stops the loop and the
|
|
// JSONL backend flushes; then drop the on-disk session log.
|
|
await ctx?.fiber.dispose()
|
|
ctx = undefined
|
|
if (root !== undefined) await rm(root, { recursive: true, force: true })
|
|
root = undefined
|
|
})
|
|
|
|
describe.skipIf(!process.env.DEEPSEEK_API_KEY)('resume: continue a persisted session across processes', () => {
|
|
it('recalls a fact stored in a prior, separately-disposed session', async () => {
|
|
root = await mkdtemp(join(tmpdir(), 'dsh-resume-e2e-'))
|
|
|
|
// Run 1: a fresh agent on a KNOWN session id learns a secret, then we
|
|
// dispose the whole context (simulating process exit) so only the JSONL
|
|
// log on disk survives.
|
|
ctx = await codingHarness(process.cwd(), { persistenceRoot: root })
|
|
const first = ctx.agents.create({
|
|
agentId: AgentId('resume-1'),
|
|
sessionId: SESSION_ID,
|
|
agentOptions: { model: 'deepseek-v4-flash', systemPrompt: SYSTEM_PROMPT },
|
|
}).agent as ReactLoopAgent
|
|
first.send([{ type: 'text', text: `Remember this code for later: ${SECRET}. Just acknowledge it.` }])
|
|
await waitForIdle(ctx, first)
|
|
await ctx.fiber.dispose()
|
|
ctx = undefined
|
|
|
|
// Run 2: a brand-new context over the SAME root resumes the persisted
|
|
// session. The loaded event log seeds the live session, so the model sees
|
|
// run 1's exchange as conversation history.
|
|
ctx = await codingHarness(process.cwd(), { persistenceRoot: root })
|
|
const resumed = (await ctx.agents.resume({
|
|
agentId: AgentId('resume-2'),
|
|
resumeSessionId: SESSION_ID,
|
|
agentOptions: { model: 'deepseek-v4-flash', systemPrompt: SYSTEM_PROMPT },
|
|
})).agent as ReactLoopAgent
|
|
expect(resumed.session.id).toBe(SESSION_ID)
|
|
// The prior user turn is in the rehydrated log before the model is asked.
|
|
expect(JSON.stringify(resumed.session.deriveMessages())).toContain(SECRET)
|
|
|
|
resumed.send([{ type: 'text', text: 'What was the code I asked you to remember? Reply with just the code.' }])
|
|
await waitForIdle(ctx, resumed)
|
|
|
|
// The model recalls it — only possible from the resumed history.
|
|
expect(finalText([...resumed.session.events])).toContain(SECRET)
|
|
}, 180_000)
|
|
})
|