mirror of
https://github.com/deepseek-ai/deepseek-harness
synced 2026-08-15 21:04:50 +00:00
Merge branch 'codex/simp-prune-tools-prompt-surface' into codex/simp-prune-code-runtime-surface
# Conflicts: # docs/config-catalog.md # docs/cordis-catalog/services.md # docs/rfc/implemented/feature/2026-06-15-code-mode.md
This commit is contained in:
@@ -1,15 +1,7 @@
|
||||
/**
|
||||
* Code Mode: the `run_code` tool and its dispatch bridge. The model writes a
|
||||
* TypeScript program; the bridge hands it to `ctx.codeRuntime` with one async
|
||||
* binding per end capability visible to the calling agent, then serializes
|
||||
* every binding call through a per-run queue onto `ToolRegistry.execute()`.
|
||||
* Sub-calls therefore traverse the complete pre/guard/around/post/final-result
|
||||
* pipeline exactly like native calls and carry the outer execution's opaque
|
||||
* token for correlation. The bridge logs each sub-dispatch as a
|
||||
* `tool/code-dispatch` session event and returns only the program's curated
|
||||
* output. The registry itself decides WHEN this tool exists (its `mode`
|
||||
* config); this module owns only the tool and the bridge.
|
||||
*
|
||||
* Code Mode `run_code` transport. Programs call the registry's agent-visible
|
||||
* tools through nested, sequential executions; each sub-dispatch is logged for
|
||||
* reconstruction, while only the outer curated result enters model history.
|
||||
* @module @deepseek-ai/dsh-tools/src/code-mode
|
||||
*/
|
||||
|
||||
@@ -24,16 +16,11 @@ import type { ToolDefinition, ToolRegistry } from './index.ts'
|
||||
declare module '@deepseek-ai/dsh-session' {
|
||||
interface SessionEventMap {
|
||||
/**
|
||||
* One bridged sub-dispatch from a `run_code` program: the parent
|
||||
* `run_code` call id, the deterministic sub-call id
|
||||
* (`<parent>:code:<n>`), the tool `name` with its JSON-normalized
|
||||
* `arguments` — the exact value dispatched, normalized BEFORE dispatch,
|
||||
* so this append can never fail on payload shape — whether the sub-call
|
||||
* errored, and a bounded `resultSummary` of its model-facing text.
|
||||
* Log-only: `deriveMessages()` ignores it, so sub-calls never re-enter
|
||||
* model context; persistence and UIs get every call. Appended inside the
|
||||
* parent `run_code`'s execution (the bridge drains its queue before
|
||||
* returning), so the turn-enclosure invariant holds by construction.
|
||||
* One bridged sub-dispatch from a `run_code` program: the parent `run_code` call id, the
|
||||
* deterministic sub-call id (`<parent>:code:<n>`), the tool `name` with its
|
||||
* JSON-normalized `arguments` — the exact value dispatched, normalized before dispatch, so
|
||||
* this append can never fail on payload shape — whether the sub-call errored, and a
|
||||
* bounded `resultSummary` of its model-facing text.
|
||||
*/
|
||||
'tool/code-dispatch': { parentCallId: CallId; subCallId: CallId; name: string; arguments: unknown; isError: boolean; resultSummary: string }
|
||||
}
|
||||
@@ -89,16 +76,11 @@ function summarize(text: string): string {
|
||||
}
|
||||
|
||||
/**
|
||||
* JSON-normalize one binding call's argument into TWO independent parses of
|
||||
* the same canonical text: `dispatched` goes to the tool, `logged` to the
|
||||
* `tool/code-dispatch` event — identical by construction (the runtime's
|
||||
* structured-clone boundary is wider than JSON; the session log accepts only
|
||||
* JSON), and separate objects, so a tool mutating its args can neither
|
||||
* desync the log from what was dispatched nor re-poison the append. A value
|
||||
* that does not survive the round-trip (`undefined` — the log rejects it as
|
||||
* event data — `BigInt`, a circular structure, a bare function) rejects that
|
||||
* one call BEFORE dispatch with a model-correctable error: nothing ever
|
||||
* executes unlogged.
|
||||
* JSON-normalize one binding call's argument into TWO independent parses of the same canonical
|
||||
* text: `dispatched` goes to the tool, `logged` to the `tool/code-dispatch` event — identical
|
||||
* by construction (the runtime's structured-clone boundary is wider than JSON; the session log
|
||||
* accepts only JSON), and separate objects, so a tool mutating its args can neither desync the
|
||||
* log from what was dispatched nor re-poison the append.
|
||||
*/
|
||||
function jsonNormalizeArgs(value: unknown): { dispatched: unknown; logged: unknown } {
|
||||
if (value === undefined) {
|
||||
@@ -138,7 +120,7 @@ function asRunCodeMeta(meta: unknown): RunCodeMeta | undefined {
|
||||
|
||||
/**
|
||||
* Build the `run_code` {@link ToolDefinition}: one required `code` parameter,
|
||||
* executed through the dispatch bridge described in the module doc. The
|
||||
* executed through the dispatch bridge described above. The
|
||||
* registry reserves it as presentation infrastructure under non-native modes,
|
||||
* outside the filterable global/scoped capability layers.
|
||||
* @param registry - the owning registry (sub-calls go through its `execute`,
|
||||
@@ -171,11 +153,9 @@ export function createRunCodeTool(registry: ToolRegistry, requireRuntime: () =>
|
||||
exec.signal?.addEventListener('abort', onOuterAbort, { once: true })
|
||||
|
||||
let dispatches = 0
|
||||
// The per-run serialization queue: every binding call chains onto the
|
||||
// tail, so even `Promise.all` executes the underlying tool calls one at
|
||||
// a time in submission order (the tool contract carries no
|
||||
// concurrency-safety metadata yet). The fold keeps the tail non-rejecting
|
||||
// so one failed dispatch never poisons the chain.
|
||||
// The per-run serialization queue: every binding call chains onto the tail, so even
|
||||
// `Promise.all` executes the underlying tool calls one at a time in submission order (the
|
||||
// tool contract carries no concurrency-safety metadata yet).
|
||||
let queue: Promise<void> = Promise.resolve()
|
||||
const enqueue = <T>(task: () => Promise<T>): Promise<T> => {
|
||||
const turn = queue.then(() => {
|
||||
@@ -210,11 +190,9 @@ export function createRunCodeTool(registry: ToolRegistry, requireRuntime: () =>
|
||||
signal: runController.signal,
|
||||
})
|
||||
const text = textOf(result.content)
|
||||
// Sub-call `additionalContext` is deliberately DROPPED here: the
|
||||
// loop's buffering (append after the step's tool/results) has no
|
||||
// safe analogue from inside a running run_code — injecting now
|
||||
// would break tool-call/result adjacency. Deferred until a real
|
||||
// hook needs it through Code Mode.
|
||||
// Sub-call `additionalContext` is deliberately DROPPED here: the loop's buffering
|
||||
// (append after the step's tool/results) has no safe analogue from inside a running
|
||||
// run_code — injecting now would break tool-call/result adjacency.
|
||||
exec.agent?.session.append('tool/code-dispatch', {
|
||||
parentCallId: exec.callId,
|
||||
subCallId,
|
||||
@@ -265,18 +243,8 @@ export function createRunCodeTool(registry: ToolRegistry, requireRuntime: () =>
|
||||
signal: runController.signal,
|
||||
})
|
||||
} finally {
|
||||
// Quiescence before returning, whether the runtime fulfilled or
|
||||
// REJECTED (a backend that starts a binding call and then throws
|
||||
// must not leak a live sub-dispatch past this settlement): fire
|
||||
// the run-scoped abort (cancelling an in-flight sub-dispatch,
|
||||
// abandoning queued ones), then await the queue's drain — an
|
||||
// aborted sub-call still settles and logs its event INSIDE the
|
||||
// open turn; nothing can append after we return. `queue` is the
|
||||
// FOLDED tail (every link swallows its rejection into undefined),
|
||||
// so this await cannot itself reject — an abandoned queued call
|
||||
// can never mask the runtime's own failure, returned or thrown;
|
||||
// rejections surface only on the per-call promises the program
|
||||
// holds.
|
||||
// Abort sub-dispatches and drain the folded queue before closing the turn.
|
||||
// Binding failures remain observable through their individual promises.
|
||||
runController.abort('run_code settled')
|
||||
await queue
|
||||
}
|
||||
@@ -296,13 +264,7 @@ export function createRunCodeTool(registry: ToolRegistry, requireRuntime: () =>
|
||||
exec.signal?.removeEventListener('abort', onOuterAbort)
|
||||
}
|
||||
},
|
||||
// The program IS the title, the way command tools title their cards with
|
||||
// the command: an execute-card's title is the one slot an ACP client
|
||||
// always shows (Zed's execute cards render no body content and no raw
|
||||
// input without a real terminal attached), so anywhere else the code
|
||||
// would be invisible. Multi-line titles are the execute-card idiom —
|
||||
// capable clients render them whole; others truncate to the first line
|
||||
// and still hold the full program in rawInput.
|
||||
// ACP execute cards use the program as their visible title.
|
||||
presentCall: args => ({
|
||||
card: 'generic',
|
||||
title: args.code,
|
||||
|
||||
Reference in New Issue
Block a user