mirror of
https://github.com/deepseek-ai/deepseek-harness
synced 2026-08-15 21:04:50 +00:00
Merge remote-tracking branch 'origin/master' into codex/skill-system
# Conflicts: # docs/architecture.md # docs/config-catalog.md # docs/module-graph.md # docs/rfc/INDEX.md # examples/acp-agent/tests/snapshots/text-turn/session.jsonl # packages/core/agent-core/src/index.ts # packages/core/tools/tests/gen-tool-catalog.spec.ts # packages/support/acp-snapshot/src/suite.ts # packages/ui/acp-agent/src/index.ts
This commit is contained in:
@@ -1,9 +1,18 @@
|
||||
# dsh-tools
|
||||
|
||||
Tool registry and execution pipeline. Tool plugins register their schemas and executors; the agent loop executes each call through `tools/pre-execute` (the allow/deny gate) → `tools/execute` (an around-dispatch wrapper for timeout/retry/metrics plugins) → `tools/post-execute` (inspect/replace the result, attach context).
|
||||
Tool registry and execution pipeline. Tool plugins register their schemas and executors; the agent loop executes each call through `tools/pre-execute` (the allow/deny gate) → `tools/execute` (an around-dispatch wrapper for timeout/retry/metrics plugins) → `tools/post-execute` (inspect/replace the result, attach context). The registry also owns HOW its tools are presented to the model — its `mode` config selects native function calling, [Code Mode](#code-mode), or both.
|
||||
|
||||
## Service: `ToolRegistry` (ctx key: `tools`)
|
||||
|
||||
### Config
|
||||
|
||||
```yaml
|
||||
tools:
|
||||
mode: native # native (default) | code | both
|
||||
```
|
||||
|
||||
`native` contributes every registered tool as a wire function definition — the default, byte-for-byte the pre-config behavior. `code` contributes exactly ONE wire tool, `run_code`, plus the generated `tools:sdk` prompt section (see [Code Mode](#code-mode)). `both` contributes every native definition AND `run_code` + the SDK section. Non-native modes require a loaded `ctx.codeRuntime` with `language: 'typescript'`; a missing or mismatched runtime rejects every prompt assembly with an actionable error, and a `systemPrompt.toolOrder` naming tools the mode no longer contributes rejects the assembly the same way.
|
||||
|
||||
### Public API
|
||||
|
||||
- `ctx.tools.register(definition: ToolDefinition): () => void` Register a tool. Disposed with the calling fiber.
|
||||
@@ -119,6 +128,16 @@ const bash = defineTool({
|
||||
})
|
||||
```
|
||||
|
||||
### Code Mode
|
||||
|
||||
Under `mode: code` (or `both`) the registry turns the tool surface into a programming API, per the [Code Mode RFC](../../../docs/rfc/implemented/feature/2026-06-15-code-mode.md): the model writes a TypeScript program (the body of an async function) and passes it to the ONE wire tool `run_code`; the program runs in `ctx.codeRuntime` (the [code-execution seam](../../code-runtime/README.md) — the shipped backend is a worker thread) with one async binding per registered tool (`await tools.bash({...})`), and ONLY what it prints or returns re-enters the model's context.
|
||||
|
||||
- **The SDK section** (`tools:sdk`, order 150): a lazy prompt section regenerating, at each assembly, a `declare const tools: {...}` TypeScript declaration of every registered tool except `run_code` (exotic names via quoted keys), plus fixed usage instructions. Deterministic — lexicographic tool order, byte-identical text for an unchanged tool set (prefix-cache-friendly). The codegen (`jsonSchemaToTs`, exported) is TOTAL: constructs outside the `defineTool` subset degrade to `unknown`, never throw.
|
||||
- **The dispatch bridge** (`run_code`'s execute): every binding call is JSON-normalized BEFORE dispatch (a value that does not survive — `BigInt`, circulars — rejects that one call, so the dispatched form and the logged form are the same JSON value by construction), serialized through a per-run queue (even `Promise.all` executes the underlying `ctx.tools.execute()` calls one at a time in submission order — the tool contract carries no concurrency-safety metadata yet), gated by `tools/pre-execute`/`tools/post-execute` like any native call (a deny reaches the program as a binding rejection), and logged as one `tool/code-dispatch` session event (log-only: `deriveMessages()` never surfaces it) with the deterministic sub-id `<parent>:code:<n>`. A failed sub-call REJECTS the program-side promise with the tool's error text — real code error handling, no bespoke envelope. A sub-call's `additionalContext` is deliberately DROPPED (no safe outlet mid-run without breaking tool-call/result adjacency; deferred until a real hook needs it through Code Mode).
|
||||
- **Settlement discipline**: the bridge owns a run-scoped abort that follows the outer signal in and fires when the run settles for any reason, so a budget expiry aborts an in-flight sub-tool instead of orphaning it; the bridge then drains its queue BEFORE returning, so every `tool/code-dispatch` lands inside the open turn. A failed run throws `CodeRunFailedError` (`code: 'CODE_RUN_FAILED'`, message = the failure kind + captured logs), which the pipeline converts to a structured `isError` the model self-corrects from.
|
||||
|
||||
The wire collapse is the registry's own contribution (`systemPrompt.tools()` is mode-aware), so the logged `request/header` records it for free — under `code`, the assembled tool list is exactly `[run_code]`, pinned by tests and the snapshot goldens. Try it: `pnpm run demo:code-mode` ([the coding-agent example's Code Mode overlay](../../../examples/coding-agent/README.md#code-mode)); `pnpm run demo:code-mode acp` serves the same mode over ACP instead of the REPL.
|
||||
|
||||
### What is NOT here (TODO)
|
||||
|
||||
- **Tool shapes review** — when real tools land (e.g. a concurrency-safety hint for parallel execution); phase 1 executes tool calls sequentially.
|
||||
|
||||
@@ -23,13 +23,20 @@
|
||||
"license": "BSD-3-Clause",
|
||||
"peerDependencies": {
|
||||
"@deepseek-ai/dsh-agent": "^0.0.1",
|
||||
"@deepseek-ai/dsh-code-runtime": "^0.0.1",
|
||||
"@deepseek-ai/dsh-llm": "^0.0.1",
|
||||
"@deepseek-ai/dsh-session": "^0.0.1",
|
||||
"@deepseek-ai/dsh-system-prompt": "^0.0.1",
|
||||
"cordis": "^4.0.0-rc.6"
|
||||
},
|
||||
"dependencies": {
|
||||
"schemastery": "^3.18.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@deepseek-ai/dsh-agent": "workspace:^",
|
||||
"@deepseek-ai/dsh-code-runtime": "workspace:^",
|
||||
"@deepseek-ai/dsh-llm": "workspace:^",
|
||||
"@deepseek-ai/dsh-session": "workspace:^",
|
||||
"@deepseek-ai/dsh-system-prompt": "workspace:^",
|
||||
"cordis": "^4.0.0-rc.6"
|
||||
}
|
||||
|
||||
318
packages/core/tools/src/code-mode.ts
Normal file
318
packages/core/tools/src/code-mode.ts
Normal file
@@ -0,0 +1,318 @@
|
||||
/**
|
||||
* Code Mode: the `run_code` tool and its dispatch bridge. The model writes a
|
||||
* TypeScript program; the bridge hands it to `ctx.codeRuntime` with one
|
||||
* async binding per registered tool, serializes every binding call through a
|
||||
* per-run queue onto `ToolRegistry.execute()` (so `tools/pre-execute` /
|
||||
* `tools/post-execute` gate sub-calls exactly like native ones), logs each
|
||||
* sub-dispatch as a `tool/code-dispatch` session event, and returns only the
|
||||
* program's curated output. The registry itself decides WHEN this tool
|
||||
* exists (its `mode` config); this module owns only the tool and the bridge.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-tools/src/code-mode
|
||||
*/
|
||||
|
||||
import { inspect } from 'node:util'
|
||||
import { CallId, HarnessError } from '@deepseek-ai/dsh-llm'
|
||||
import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
||||
import type { CodeBindingFunction, CodeRunResult, CodeRuntime } from '@deepseek-ai/dsh-code-runtime'
|
||||
import type {} from '@deepseek-ai/dsh-session'
|
||||
import { defineTool } from './schema.ts'
|
||||
import type { ToolDefinition, ToolRegistry } from './index.ts'
|
||||
|
||||
declare module '@deepseek-ai/dsh-session' {
|
||||
interface SessionEventMap {
|
||||
/**
|
||||
* One bridged sub-dispatch from a `run_code` program: the parent
|
||||
* `run_code` call id, the deterministic sub-call id
|
||||
* (`<parent>:code:<n>`), the tool `name` with its JSON-normalized
|
||||
* `arguments` — the exact value dispatched, normalized BEFORE dispatch,
|
||||
* so this append can never fail on payload shape — whether the sub-call
|
||||
* errored, and a bounded `resultSummary` of its model-facing text.
|
||||
* Log-only: `deriveMessages()` ignores it, so sub-calls never re-enter
|
||||
* model context; persistence and UIs get every call. Appended inside the
|
||||
* parent `run_code`'s execution (the bridge drains its queue before
|
||||
* returning), so the turn-enclosure invariant holds by construction.
|
||||
*/
|
||||
'tool/code-dispatch': { parentCallId: CallId; subCallId: CallId; name: string; arguments: unknown; isError: boolean; resultSummary: string }
|
||||
}
|
||||
}
|
||||
|
||||
/** The model-facing name of the Code Mode tool. */
|
||||
export const RUN_CODE_NAME = 'run_code'
|
||||
|
||||
/** The `tools:sdk` section order: inside the 100–199 tool-guidance band, after per-tool guidance sections. */
|
||||
export const SDK_SECTION_ORDER = 150
|
||||
|
||||
/**
|
||||
* Thrown by `run_code` when the program run itself failed — a program
|
||||
* exception, a budget expiry, an abort, or substrate death. Extends
|
||||
* {@link HarnessError} (`code: 'CODE_RUN_FAILED'`); the registry's execution
|
||||
* pipeline converts it into a structured `isError` result whose text carries
|
||||
* the failure kind plus the captured logs, so the model can self-correct.
|
||||
*/
|
||||
export class CodeRunFailedError extends HarnessError {
|
||||
constructor(message: string) {
|
||||
super(message, 'CODE_RUN_FAILED')
|
||||
this.name = 'CodeRunFailedError'
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Cap for a `tool/code-dispatch` event's `resultSummary`. A log-ergonomics
|
||||
* constant, not config: the full result already flows to the program; the
|
||||
* summary exists so log readers see what a sub-call returned at a glance.
|
||||
*/
|
||||
const SUMMARY_MAX_CHARS = 200
|
||||
|
||||
/** Bounded inspect for rendering a program's completion value into the model-facing text. */
|
||||
const INSPECT_OPTIONS = { depth: 4, maxArrayLength: 100, maxStringLength: 10_000 } as const
|
||||
|
||||
/** Join a result's text blocks; a non-text block becomes a placeholder (an MVP limitation, stated in the SDK instructions). */
|
||||
function textOf(content: ContentBlock[]): string {
|
||||
return content
|
||||
.map((block) => {
|
||||
switch (block.type) {
|
||||
case 'text': return block.text
|
||||
// ContentBlockMap is merge-extensible — future block kinds land here
|
||||
// deliberately (no assertNever on merge-extensible unions).
|
||||
default: return `[${block.type} content]`
|
||||
}
|
||||
})
|
||||
.join('\n')
|
||||
}
|
||||
|
||||
/** Bound a sub-call's model-facing text for the log event's `resultSummary`. */
|
||||
function summarize(text: string): string {
|
||||
return text.length > SUMMARY_MAX_CHARS ? `${text.slice(0, SUMMARY_MAX_CHARS)}…` : text
|
||||
}
|
||||
|
||||
/**
|
||||
* JSON-normalize one binding call's argument into TWO independent parses of
|
||||
* the same canonical text: `dispatched` goes to the tool, `logged` to the
|
||||
* `tool/code-dispatch` event — identical by construction (the runtime's
|
||||
* structured-clone boundary is wider than JSON; the session log accepts only
|
||||
* JSON), and separate objects, so a tool mutating its args can neither
|
||||
* desync the log from what was dispatched nor re-poison the append. A value
|
||||
* that does not survive the round-trip (`undefined` — the log rejects it as
|
||||
* event data — `BigInt`, a circular structure, a bare function) rejects that
|
||||
* one call BEFORE dispatch with a model-correctable error: nothing ever
|
||||
* executes unlogged.
|
||||
*/
|
||||
function jsonNormalizeArgs(value: unknown): { dispatched: unknown; logged: unknown } {
|
||||
if (value === undefined) {
|
||||
throw new Error('tool arguments must be JSON-serializable (call the tool with an arguments object, e.g. `{}`)')
|
||||
}
|
||||
let text: string | undefined
|
||||
try {
|
||||
text = JSON.stringify(value)
|
||||
} catch (error: unknown) {
|
||||
throw new Error(`tool arguments must be JSON-serializable: ${error instanceof Error ? error.message : String(error)}`)
|
||||
}
|
||||
// JSON.stringify's lib type claims `string`, but a bare function or symbol
|
||||
// root really yields `undefined` at runtime — the guard is live.
|
||||
// eslint-disable-next-line @typescript-eslint/no-unnecessary-condition
|
||||
if (text === undefined) throw new Error('tool arguments must be JSON-serializable (got a value JSON cannot represent)')
|
||||
return { dispatched: JSON.parse(text) as unknown, logged: JSON.parse(text) as unknown }
|
||||
}
|
||||
|
||||
/** Render the program's completion value for the model-facing result text (`''` when the program returned nothing). */
|
||||
function renderValue(value: unknown): string {
|
||||
if (value === undefined) return ''
|
||||
return typeof value === 'string' ? value : inspect(value, INSPECT_OPTIONS)
|
||||
}
|
||||
|
||||
/** The run_code result's `meta` payload (JSON-serializable; `presentResult` narrows it back). */
|
||||
interface RunCodeMeta {
|
||||
logs: CodeRunResult['logs']
|
||||
dispatches: number
|
||||
}
|
||||
|
||||
/** Soft-narrow a result `meta` back to {@link RunCodeMeta} (replay may carry older shapes; presentation must not throw). */
|
||||
function asRunCodeMeta(meta: unknown): RunCodeMeta | undefined {
|
||||
if (typeof meta !== 'object' || meta === null) return undefined
|
||||
const m = meta as Record<string, unknown>
|
||||
if (!Array.isArray(m.logs) || typeof m.dispatches !== 'number') return undefined
|
||||
return m as unknown as RunCodeMeta
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the `run_code` {@link ToolDefinition}: one required `code` parameter,
|
||||
* executed through the dispatch bridge described in the module doc. The
|
||||
* registry registers it under non-native modes.
|
||||
* @param registry - the owning registry (sub-calls go through its `execute`,
|
||||
* bindings cover its registered tools).
|
||||
* @param requireRuntime - resolves `ctx.codeRuntime` or throws the loud
|
||||
* misconfiguration error (shared with the registry's assembly-time checks).
|
||||
* @returns the registry-ready definition.
|
||||
*/
|
||||
export function createRunCodeTool(registry: ToolRegistry, requireRuntime: () => CodeRuntime): ToolDefinition {
|
||||
return defineTool({
|
||||
name: RUN_CODE_NAME,
|
||||
description:
|
||||
'Execute a TypeScript program against the available tools. Write the BODY of an '
|
||||
+ 'async function (erasable syntax only; top-level `await` and `return` work) and '
|
||||
+ 'call tools as `await tools.name(args)` per the declarations in the system prompt. '
|
||||
+ 'Only what you print or return comes back — curate it.',
|
||||
parameters: {
|
||||
code: { type: 'string', required: true, description: 'The program: the body of an async TypeScript function.' },
|
||||
},
|
||||
async execute(args, exec) {
|
||||
const runtime = requireRuntime()
|
||||
|
||||
// The run-scoped abort: follows the outer signal in, and fires when the
|
||||
// run settles for ANY reason, so an in-flight sub-dispatch is aborted
|
||||
// (its executor kills on this signal) instead of orphaned, and
|
||||
// queued-unstarted dispatches are abandoned.
|
||||
const runController = new AbortController()
|
||||
const onOuterAbort = (): void => { runController.abort(exec.signal?.reason) }
|
||||
if (exec.signal?.aborted) onOuterAbort()
|
||||
exec.signal?.addEventListener('abort', onOuterAbort, { once: true })
|
||||
|
||||
let dispatches = 0
|
||||
// The per-run serialization queue: every binding call chains onto the
|
||||
// tail, so even `Promise.all` executes the underlying tool calls one at
|
||||
// a time in submission order (the tool contract carries no
|
||||
// concurrency-safety metadata yet). The fold keeps the tail non-rejecting
|
||||
// so one failed dispatch never poisons the chain.
|
||||
let queue: Promise<void> = Promise.resolve()
|
||||
const enqueue = <T>(task: () => Promise<T>): Promise<T> => {
|
||||
const turn = queue.then(() => {
|
||||
if (runController.signal.aborted) {
|
||||
throw new Error(`run_code run is over (${String(runController.signal.reason)}); tool call abandoned`)
|
||||
}
|
||||
return task()
|
||||
})
|
||||
queue = turn.then(() => undefined, () => undefined)
|
||||
return turn
|
||||
}
|
||||
|
||||
// Read through a call, not a bare property: the abort state genuinely
|
||||
// changes across awaits, and a direct `.aborted` re-check after one
|
||||
// would be narrowed away by control flow analysis.
|
||||
const runOver = (): boolean => runController.signal.aborted
|
||||
|
||||
const binding = (name: string): CodeBindingFunction => async (rawArgs: unknown): Promise<unknown> => {
|
||||
if (runOver()) {
|
||||
throw new Error(`run_code run is over (${String(runController.signal.reason)}); ${name} not dispatched`)
|
||||
}
|
||||
const normalized = jsonNormalizeArgs(rawArgs)
|
||||
const outcome = await enqueue(async () => {
|
||||
const n = ++dispatches
|
||||
const subCallId = CallId(`${String(exec.callId)}:code:${n}`)
|
||||
const result = await registry.execute({
|
||||
callId: subCallId,
|
||||
name,
|
||||
arguments: normalized.dispatched,
|
||||
...exec.agent ? { agent: exec.agent } : {},
|
||||
signal: runController.signal,
|
||||
})
|
||||
const text = textOf(result.content)
|
||||
// Sub-call `additionalContext` is deliberately DROPPED here: the
|
||||
// loop's buffering (append after the step's tool/results) has no
|
||||
// safe analogue from inside a running run_code — injecting now
|
||||
// would break tool-call/result adjacency. Deferred until a real
|
||||
// hook needs it through Code Mode.
|
||||
exec.agent?.session.append('tool/code-dispatch', {
|
||||
parentCallId: exec.callId,
|
||||
subCallId,
|
||||
name,
|
||||
// The SIBLING parse of the dispatched value: byte-identical JSON,
|
||||
// but a separate object — a tool mutating its args cannot desync
|
||||
// this record from what it actually received.
|
||||
arguments: normalized.logged,
|
||||
isError: result.isError,
|
||||
resultSummary: summarize(text),
|
||||
})
|
||||
return { text, isError: result.isError }
|
||||
})
|
||||
// A budget expiry or outer cancel that lands while this call was in
|
||||
// flight already aborted the dispatch; stop the program now rather
|
||||
// than hand it a result from a run that is over.
|
||||
if (runOver()) {
|
||||
throw new Error(`run_code run is over (${String(runController.signal.reason)}); ${name} result discarded`)
|
||||
}
|
||||
// A failed tool call REJECTS — real code signals failure by throwing,
|
||||
// so try/catch and Promise.all short-circuiting behave as models
|
||||
// expect (the error text is the tool's model-facing result text).
|
||||
if (outcome.isError) throw new Error(outcome.text)
|
||||
return outcome.text
|
||||
}
|
||||
|
||||
// Null-prototype + defineProperty, mirroring the worker-side namespace
|
||||
// build: a registered tool named `__proto__` must become an ordinary
|
||||
// own key (a plain-object assignment would hit the prototype setter,
|
||||
// silently dropping the binding), and the runtime host resolves
|
||||
// binding names as own properties only.
|
||||
const functions: Record<string, CodeBindingFunction> = Object.create(null) as Record<string, CodeBindingFunction>
|
||||
for (const schema of registry.schemas()) {
|
||||
if (schema.name === RUN_CODE_NAME) continue
|
||||
Object.defineProperty(functions, schema.name, { enumerable: true, value: binding(schema.name) })
|
||||
}
|
||||
|
||||
try {
|
||||
let result: CodeRunResult
|
||||
try {
|
||||
result = await runtime.run({
|
||||
program: args.code,
|
||||
bindings: [{ global: 'tools', functions }],
|
||||
signal: runController.signal,
|
||||
})
|
||||
} finally {
|
||||
// Quiescence before returning, whether the runtime fulfilled or
|
||||
// REJECTED (a backend that starts a binding call and then throws
|
||||
// must not leak a live sub-dispatch past this settlement): fire
|
||||
// the run-scoped abort (cancelling an in-flight sub-dispatch,
|
||||
// abandoning queued ones), then await the queue's drain — an
|
||||
// aborted sub-call still settles and logs its event INSIDE the
|
||||
// open turn; nothing can append after we return. `queue` is the
|
||||
// FOLDED tail (every link swallows its rejection into undefined),
|
||||
// so this await cannot itself reject — an abandoned queued call
|
||||
// can never mask the runtime's own failure, returned or thrown;
|
||||
// rejections surface only on the per-call promises the program
|
||||
// holds.
|
||||
runController.abort('run_code settled')
|
||||
await queue
|
||||
}
|
||||
|
||||
if (result.error) {
|
||||
const logsText = result.logs.length > 0 ? `\nCaptured output:\n${result.logs.map(entry => entry.text).join('\n')}` : ''
|
||||
throw new CodeRunFailedError(`code run failed (${result.error.kind}): ${result.error.message}${logsText}`)
|
||||
}
|
||||
const rendered = renderValue(result.value)
|
||||
const parts = [result.logs.map(entry => entry.text).join('\n'), rendered].filter(part => part.length > 0)
|
||||
const meta: RunCodeMeta = { logs: result.logs, dispatches }
|
||||
return {
|
||||
content: [{ type: 'text', text: parts.length > 0 ? parts.join('\n') : '(run_code completed with no output)' }],
|
||||
meta,
|
||||
}
|
||||
} finally {
|
||||
exec.signal?.removeEventListener('abort', onOuterAbort)
|
||||
}
|
||||
},
|
||||
// The program IS the title, the way command tools title their cards with
|
||||
// the command: an execute-card's title is the one slot an ACP client
|
||||
// always shows (Zed's execute cards render no body content and no raw
|
||||
// input without a real terminal attached), so anywhere else the code
|
||||
// would be invisible. Multi-line titles are the execute-card idiom —
|
||||
// capable clients render them whole; others truncate to the first line
|
||||
// and still hold the full program in rawInput.
|
||||
presentCall: args => ({
|
||||
card: 'generic',
|
||||
title: args.code,
|
||||
kind: 'execute',
|
||||
rawInput: args.code,
|
||||
}),
|
||||
// Title omitted on the result: an update replaces only the fields it
|
||||
// carries, so the pending card's program title persists through
|
||||
// completion; the captured output rides as body content.
|
||||
presentResult: (_args, result) => {
|
||||
const meta = asRunCodeMeta(result.meta)
|
||||
if (!meta) return undefined
|
||||
const output = meta.logs.map(entry => entry.text).join('\n')
|
||||
return {
|
||||
card: 'generic',
|
||||
...output.length > 0 ? { content: [{ type: 'text' as const, text: output }] } : {},
|
||||
}
|
||||
},
|
||||
})
|
||||
}
|
||||
@@ -6,15 +6,26 @@
|
||||
* (inspect/replace the result, attach context) for sandbox, permission, and hook
|
||||
* plugins to gate or transform a call.
|
||||
*
|
||||
* The registry also owns HOW its tools are presented to the model — its
|
||||
* `mode` config: `'native'` (every tool as a wire function definition,
|
||||
* today's behavior and the default), `'code'` (the wire carries exactly one
|
||||
* tool, `run_code`, plus a generated TypeScript SDK prompt section), or
|
||||
* `'both'`. See `code-mode.ts` (the tool + dispatch bridge) and
|
||||
* `ts-types.ts` (the SDK codegen); design in the Code Mode RFC.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-tools
|
||||
*/
|
||||
|
||||
import { Context, Service } from 'cordis'
|
||||
import z from 'schemastery'
|
||||
import type { CallId, ContentBlock, ToolSchema } from '@deepseek-ai/dsh-llm'
|
||||
import { HarnessError } from '@deepseek-ai/dsh-llm'
|
||||
import type { Agent, HookContext } from '@deepseek-ai/dsh-agent'
|
||||
import type {} from '@deepseek-ai/dsh-system-prompt'
|
||||
import type { CodeRuntime } from '@deepseek-ai/dsh-code-runtime'
|
||||
import type { ToolCallView, ToolResultView } from './presentation.ts'
|
||||
import { createRunCodeTool, RUN_CODE_NAME, SDK_SECTION_ORDER } from './code-mode.ts'
|
||||
import { renderToolsSdk } from './ts-types.ts'
|
||||
|
||||
export {
|
||||
defineTool,
|
||||
@@ -39,6 +50,9 @@ export {
|
||||
type StructuredScalar,
|
||||
} from './json-schema.ts'
|
||||
|
||||
export { CodeRunFailedError, RUN_CODE_NAME } from './code-mode.ts'
|
||||
export { jsonSchemaToTs, renderToolsSdk } from './ts-types.ts'
|
||||
|
||||
// The render-intent vocabulary a tool declares via `presentCall`/`presentResult`
|
||||
// lives in its own UI-facing module; re-export it so `@deepseek-ai/dsh-tools`
|
||||
// stays the single public surface for consumers (producers + the ACP bridge).
|
||||
@@ -298,20 +312,100 @@ function errorInfo(error: unknown): ToolErrorInfo | undefined {
|
||||
return error instanceof HarnessError ? { name: error.name, code: error.code } : undefined
|
||||
}
|
||||
|
||||
/** How the registry presents its tools to the model (see {@link Config.mode}). */
|
||||
export type ToolPresentationMode = 'native' | 'code' | 'both'
|
||||
|
||||
/** Plugin config: how the registered tools are presented to the model. */
|
||||
export interface Config {
|
||||
/**
|
||||
* The presentation mode. `'native'` (the default) contributes every
|
||||
* registered tool as a wire function definition — byte-for-byte today's
|
||||
* behavior. `'code'` contributes exactly ONE wire tool, `run_code`, plus
|
||||
* the generated `tools:sdk` prompt section declaring every other tool as a
|
||||
* TypeScript API the program calls. `'both'` contributes every native
|
||||
* definition AND `run_code` + the SDK section. Non-native modes require a
|
||||
* loaded `ctx.codeRuntime` whose `language` is `'typescript'` — a missing
|
||||
* or mismatched runtime rejects every prompt assembly with an actionable
|
||||
* error (misconfiguration fails loud, before any model request). A
|
||||
* configured `systemPrompt.toolOrder` naming native tools likewise rejects
|
||||
* every assembly under `'code'` (those names are no longer contributed) —
|
||||
* a deployment switching modes updates its order config or drops it.
|
||||
*/
|
||||
mode?: ToolPresentationMode
|
||||
}
|
||||
|
||||
/**
|
||||
* Tool registry (`ctx.tools`): tool plugins register definitions; the agent
|
||||
* loop executes calls through the `tools/pre-execute` → `tools/execute` →
|
||||
* `tools/post-execute` pipeline. The registry contributes its schemas into the
|
||||
* system-prompt assembly.
|
||||
* system-prompt assembly — WHICH schemas is governed by its `mode` config
|
||||
* (see {@link Config.mode}); under a non-native mode it also registers the
|
||||
* `run_code` tool and the `tools:sdk` prompt section itself.
|
||||
*/
|
||||
export class ToolRegistry extends Service {
|
||||
static inject = ['systemPrompt']
|
||||
|
||||
private store = new Map<string, ToolDefinition>()
|
||||
static Config: z<Config> = z.object({
|
||||
mode: z.union(['native', 'code', 'both'] as const).default('native'),
|
||||
})
|
||||
|
||||
constructor(ctx: Context) {
|
||||
private store = new Map<string, ToolDefinition>()
|
||||
private readonly mode: ToolPresentationMode
|
||||
|
||||
constructor(ctx: Context, config: Config = {}) {
|
||||
super(ctx, 'tools')
|
||||
ctx.systemPrompt.tools(() => this.schemas())
|
||||
// The schema already defaulted an omitted mode; the ?? narrows the
|
||||
// optional-input type for direct (non-Loader) construction in tests.
|
||||
this.mode = config.mode ?? 'native'
|
||||
ctx.systemPrompt.tools(() => this.wireSchemas())
|
||||
if (this.mode !== 'native') {
|
||||
this.register(createRunCodeTool(this, () => this.requireCodeRuntime()))
|
||||
ctx.systemPrompt.section({
|
||||
name: 'tools:sdk',
|
||||
order: SDK_SECTION_ORDER,
|
||||
// A lazy thunk over the live store: regenerated at each assembly, in
|
||||
// lexicographic tool order, so an unchanged tool set renders
|
||||
// byte-identical text (prefix-cache-friendly) and a mid-session
|
||||
// registration surfaces exactly like a native-mode tool change.
|
||||
text: () => {
|
||||
this.requireCodeRuntime()
|
||||
return renderToolsSdk(this.schemas().filter(schema => schema.name !== RUN_CODE_NAME))
|
||||
},
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The registry's contribution to the wire tool list, per {@link Config.mode}.
|
||||
* Because `PromptAssembly.tools` is what the loop's request header
|
||||
* snapshots, the mode's collapse is logged and reconstructable for free.
|
||||
* Under a non-native mode this is also the loud misconfiguration gate: no
|
||||
* usable code runtime → every assembly rejects before any model request.
|
||||
*/
|
||||
private wireSchemas(): ToolSchema[] {
|
||||
if (this.mode === 'native') return this.schemas()
|
||||
this.requireCodeRuntime()
|
||||
const all = this.schemas()
|
||||
return this.mode === 'code' ? all.filter(schema => schema.name === RUN_CODE_NAME) : all
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the code runtime or throw the actionable misconfiguration error.
|
||||
* Read at use time (assembly / run_code execution), NOT via static
|
||||
* `inject`: an inject entry would hold `ctx.tools` — and every tool plugin
|
||||
* behind it — hostage to a code runtime existing even under `mode:
|
||||
* 'native'` (the loop's optional-backend idiom, same as
|
||||
* `sessionPersistence`).
|
||||
*/
|
||||
private requireCodeRuntime(): CodeRuntime {
|
||||
const runtime = this.ctx.get('codeRuntime')
|
||||
if (!runtime) {
|
||||
throw new Error(`dsh-tools: mode "${this.mode}" requires a code runtime — load a ctx.codeRuntime implementation (e.g. @deepseek-ai/dsh-code-runtime-worker) or set tools mode to "native"`)
|
||||
}
|
||||
if (runtime.language !== 'typescript') {
|
||||
throw new Error(`dsh-tools: mode "${this.mode}" generates a TypeScript SDK, but the loaded code runtime's language is "${runtime.language}"`)
|
||||
}
|
||||
return runtime
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
121
packages/core/tools/src/ts-types.ts
Normal file
121
packages/core/tools/src/ts-types.ts
Normal file
@@ -0,0 +1,121 @@
|
||||
/**
|
||||
* Code Mode codegen: the pure projection from registered tool schemas to the
|
||||
* TypeScript SDK text the model programs against (the `tools:sdk` prompt
|
||||
* section). Sibling of `json-schema.ts` — `schemas()` (native function
|
||||
* calling) and this module (the generated `declare const tools` surface) are
|
||||
* two projections of the same store.
|
||||
*
|
||||
* TOTAL by design: {@link jsonSchemaToTs} maps the JSON-Schema subset the
|
||||
* `defineTool` DSL emits and degrades every construct outside it (`$ref`,
|
||||
* `oneOf`, `integer`, future MCP shapes, …) to `unknown` without ever
|
||||
* throwing — codegen must never be the thing that fails an assembly.
|
||||
* Deterministic: a fixed tool set renders byte-identical text (tools in
|
||||
* lexicographic name order), so the section is prefix-cache-friendly.
|
||||
*
|
||||
* @module @deepseek-ai/dsh-tools/src/ts-types
|
||||
*/
|
||||
|
||||
import type { ToolSchema } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
/** Property names that are valid bare TS identifiers; anything else is quoted. */
|
||||
const IDENTIFIER = /^[A-Za-z_$][A-Za-z0-9_$]*$/
|
||||
|
||||
/** Render an object key: bare when it is a valid identifier, quoted otherwise (every name stays reachable, no aliasing). */
|
||||
function renderKey(name: string): string {
|
||||
return IDENTIFIER.test(name) ? name : JSON.stringify(name)
|
||||
}
|
||||
|
||||
/** One `indent`-deep line prefix (two spaces per level). */
|
||||
function pad(indent: number): string {
|
||||
return ' '.repeat(indent)
|
||||
}
|
||||
|
||||
/** A one-line JSDoc block for a schema `description`, or no lines when there is none. */
|
||||
function docLines(description: unknown, indent: number): string[] {
|
||||
if (typeof description !== 'string' || description.length === 0) return []
|
||||
// Keep the doc a single-line comment per property: descriptions are prose
|
||||
// (possibly with newlines); collapse whitespace so the rendered SDK stays
|
||||
// stable and compact. A comment-closer inside the description is escaped so
|
||||
// it cannot terminate the generated JSDoc early.
|
||||
const collapsed = description.replace(/\s+/g, ' ').trim()
|
||||
return [`${pad(indent)}/** ${collapsed.replaceAll('*/', String.raw`*\/`)} */`]
|
||||
}
|
||||
|
||||
/**
|
||||
* Map one JSON-Schema node to a TypeScript type literal. Handles exactly the
|
||||
* subset the `defineTool` DSL emits — `object` (`properties` + `required`),
|
||||
* `string` (with `enum` → a literal union), `number`, `boolean`, `array`
|
||||
* (`items`) — and returns `unknown` for anything else, without throwing.
|
||||
* @param schema - the JSON-Schema node (any shape; hostile inputs degrade).
|
||||
* @param indent - the indentation level for nested object members.
|
||||
* @returns the TS type text (multi-line for objects with properties).
|
||||
*/
|
||||
export function jsonSchemaToTs(schema: unknown, indent = 0): string {
|
||||
if (typeof schema !== 'object' || schema === null) return 'unknown'
|
||||
const node = schema as Record<string, unknown>
|
||||
switch (node.type) {
|
||||
case 'string': {
|
||||
if (Array.isArray(node.enum) && node.enum.length > 0 && node.enum.every(value => typeof value === 'string')) {
|
||||
return node.enum.map(value => JSON.stringify(value)).join(' | ')
|
||||
}
|
||||
return 'string'
|
||||
}
|
||||
case 'number': return 'number'
|
||||
case 'boolean': return 'boolean'
|
||||
case 'array': {
|
||||
const item = jsonSchemaToTs(node.items, indent)
|
||||
// Parenthesize a union item type so `('a' | 'b')[]` parses as intended.
|
||||
return item.includes('|') ? `(${item})[]` : `${item}[]`
|
||||
}
|
||||
case 'object': {
|
||||
const properties = node.properties
|
||||
if (typeof properties !== 'object' || properties === null) return 'Record<string, unknown>'
|
||||
const entries = Object.entries(properties as Record<string, unknown>)
|
||||
if (entries.length === 0) return 'Record<string, unknown>'
|
||||
const required = new Set(Array.isArray(node.required) ? node.required.filter(name => typeof name === 'string') : [])
|
||||
const lines: string[] = ['{']
|
||||
for (const [name, prop] of entries) {
|
||||
const description = typeof prop === 'object' && prop !== null ? (prop as Record<string, unknown>).description : undefined
|
||||
lines.push(...docLines(description, indent + 1))
|
||||
lines.push(`${pad(indent + 1)}${renderKey(name)}${required.has(name) ? '' : '?'}: ${jsonSchemaToTs(prop, indent + 1)};`)
|
||||
}
|
||||
lines.push(`${pad(indent)}}`)
|
||||
return lines.join('\n')
|
||||
}
|
||||
default: return 'unknown'
|
||||
}
|
||||
}
|
||||
|
||||
/** The fixed model-facing usage contract rendered above the declarations (see the Code Mode RFC's "What the model sees"). */
|
||||
const SDK_INSTRUCTIONS = `## Writing code for run_code
|
||||
|
||||
Pass \`run_code\` the body of an async TypeScript function (erasable syntax only — no \`enum\` or namespaces; type annotations are advisory, the code runs type-stripped). Inside the program:
|
||||
|
||||
- Call tools as \`await tools.name(args)\` — quoted access for exotic names: \`tools["my-tool"](args)\`. Every call resolves to the tool's text output as a string. Tool arguments must be JSON-serializable.
|
||||
- A FAILED tool call rejects with an \`Error\` carrying the tool's error text — \`try/catch\` it to handle and continue.
|
||||
- Calls execute sequentially, even under \`Promise.all\`.
|
||||
- Emit results with \`return\` and/or \`console.log(...)\`. ONLY what you print or return comes back to you — intermediate tool results never enter the conversation, so extract just what you need.
|
||||
|
||||
The available tools:`
|
||||
|
||||
/**
|
||||
* Render the full `tools:sdk` prompt section: the fixed usage instructions
|
||||
* plus one `declare const tools` interface covering every given tool.
|
||||
* Deterministic — tools are emitted in lexicographic name order, so an
|
||||
* unchanged tool set produces byte-identical text across assemblies.
|
||||
* @param schemas - the tool schemas to declare (the caller excludes
|
||||
* `run_code` itself).
|
||||
* @returns the complete section text.
|
||||
*/
|
||||
export function renderToolsSdk(schemas: ToolSchema[]): string {
|
||||
const sorted = [...schemas].sort((a, b) => a.name < b.name ? -1 : a.name > b.name ? 1 : 0)
|
||||
const members: string[] = []
|
||||
for (const schema of sorted) {
|
||||
members.push(...docLines(schema.description, 1))
|
||||
members.push(`${pad(1)}${renderKey(schema.name)}(args: ${jsonSchemaToTs(schema.parameters, 1)}): Promise<string>;`)
|
||||
}
|
||||
const declaration = members.length > 0
|
||||
? `declare const tools: {\n${members.join('\n')}\n}`
|
||||
: 'declare const tools: {}'
|
||||
return `${SDK_INSTRUCTIONS}\n\n\`\`\`ts\n${declaration}\n\`\`\``
|
||||
}
|
||||
640
packages/core/tools/tests/code-mode.spec.ts
Normal file
640
packages/core/tools/tests/code-mode.spec.ts
Normal file
@@ -0,0 +1,640 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { Context } from 'cordis'
|
||||
import { CallId } from '@deepseek-ai/dsh-llm'
|
||||
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
|
||||
import { CodeRuntime } from '@deepseek-ai/dsh-code-runtime'
|
||||
import type { CodeRunRequest, CodeRunResult } from '@deepseek-ai/dsh-code-runtime'
|
||||
import ToolRegistry, { CodeRunFailedError, RUN_CODE_NAME, defineTool } from '@deepseek-ai/dsh-tools'
|
||||
import type { Config, PostToolDecision, ToolExecutionResult } from '@deepseek-ai/dsh-tools'
|
||||
import type { Agent } from '@deepseek-ai/dsh-agent'
|
||||
import { Session, SessionId } from '@deepseek-ai/dsh-session'
|
||||
import type { SessionEventMap } from '@deepseek-ai/dsh-session'
|
||||
|
||||
/**
|
||||
* Code Mode unit tier (per the RFC's plan): provider contribution per mode,
|
||||
* misconfiguration rejections, the run_code dispatch bridge (serialization,
|
||||
* abort, JSON normalization, error mapping, events, quiescence), and HMR
|
||||
* safety — all against an in-repo fake runtime, exactly the
|
||||
* interface/implementation/consumer shape the seam promises.
|
||||
*/
|
||||
|
||||
/** A scriptable in-repo CodeRuntime: each test sets `behavior` to drive the bindings however it needs. */
|
||||
class FakeRuntime extends CodeRuntime {
|
||||
readonly language: string
|
||||
readonly isolation = 'fake'
|
||||
behavior: (request: CodeRunRequest) => Promise<CodeRunResult> = () => Promise.resolve({ logs: [] })
|
||||
lastRequest?: CodeRunRequest
|
||||
|
||||
constructor(ctx: Context, config: { language?: string } = {}) {
|
||||
super(ctx)
|
||||
this.language = config.language ?? 'typescript'
|
||||
}
|
||||
|
||||
run(request: CodeRunRequest): Promise<CodeRunResult> {
|
||||
this.lastRequest = request
|
||||
return this.behavior(request)
|
||||
}
|
||||
}
|
||||
|
||||
interface SetupOptions {
|
||||
mode?: Config['mode']
|
||||
runtime?: false | { language?: string }
|
||||
toolOrder?: string[]
|
||||
}
|
||||
|
||||
async function setup(options: SetupOptions = {}) {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt, { ...options.toolOrder ? { toolOrder: options.toolOrder } : {} })
|
||||
await ctx.plugin(ToolRegistry, { mode: options.mode ?? 'code' })
|
||||
let runtime: FakeRuntime | undefined
|
||||
if (options.runtime !== false) {
|
||||
await ctx.plugin(FakeRuntime, options.runtime ?? {})
|
||||
runtime = ctx.codeRuntime as FakeRuntime
|
||||
}
|
||||
return { ctx, tools: ctx.tools, systemPrompt: ctx.systemPrompt, runtime: runtime! }
|
||||
}
|
||||
|
||||
/** Register a trivial echo tool; returns the calls it received. */
|
||||
function registerEcho(ctx: Context, name = 'echo'): unknown[] {
|
||||
const calls: unknown[] = []
|
||||
ctx.tools.register(defineTool({
|
||||
name,
|
||||
description: `Echo tool ${name}.`,
|
||||
parameters: { value: { type: 'string', required: true } },
|
||||
execute(args) {
|
||||
calls.push(args)
|
||||
return Promise.resolve([{ type: 'text' as const, text: `${name}:${args.value}` }])
|
||||
},
|
||||
}))
|
||||
return calls
|
||||
}
|
||||
|
||||
/** A structural fake of the owning agent: captures session appends. */
|
||||
function fakeAgent(): { agent: Agent; events: { type: string; data: unknown }[] } {
|
||||
const events: { type: string; data: unknown }[] = []
|
||||
const agent = {
|
||||
session: {
|
||||
append: (type: string, data: unknown) => { events.push({ type, data }) },
|
||||
},
|
||||
} as unknown as Agent
|
||||
return { agent, events }
|
||||
}
|
||||
|
||||
/** Dispatch run_code through the registry pipeline, as the loop would. */
|
||||
async function runCode(ctx: Context, code: string, extras: { agent?: Agent; signal?: AbortSignal } = {}): Promise<ToolExecutionResult> {
|
||||
return ctx.tools.execute({
|
||||
callId: CallId('call-1'),
|
||||
name: RUN_CODE_NAME,
|
||||
arguments: { code },
|
||||
...extras.agent ? { agent: extras.agent } : {},
|
||||
...extras.signal ? { signal: extras.signal } : {},
|
||||
})
|
||||
}
|
||||
|
||||
describe('mode-aware wire contribution', () => {
|
||||
it("mode 'native' contributes every schema, no run_code, no SDK section — and needs no runtime", async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'native', runtime: false })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
expect(assembly.tools.map(tool => tool.name)).toEqual(['echo'])
|
||||
expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
|
||||
})
|
||||
|
||||
it("mode 'code' contributes exactly [run_code] plus the SDK section declaring the other tools", async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'code' })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
expect(assembly.tools.map(tool => tool.name)).toEqual([RUN_CODE_NAME])
|
||||
const sdk = assembly.sections.find(section => section.name === 'tools:sdk')
|
||||
expect(sdk?.text).toContain('declare const tools: {')
|
||||
expect(sdk?.text).toContain('echo(args:')
|
||||
expect(sdk?.text).not.toContain('run_code(args:')
|
||||
})
|
||||
|
||||
it("mode 'both' contributes every native schema plus run_code, and the SDK section", async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'both' })
|
||||
registerEcho(ctx)
|
||||
const assembly = await systemPrompt.assemble()
|
||||
expect(assembly.tools.map(tool => tool.name)).toEqual(['echo', RUN_CODE_NAME])
|
||||
expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(true)
|
||||
})
|
||||
|
||||
it("never exposes run_code to programs, even under mode 'both' (no recursive dispatch path)", async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'both' })
|
||||
registerEcho(ctx)
|
||||
runtime.behavior = (request) => {
|
||||
const functions = request.bindings[0]!.functions
|
||||
return Promise.resolve({
|
||||
logs: [],
|
||||
value: JSON.stringify({
|
||||
names: Object.keys(functions).sort(),
|
||||
// Own-property AND prototype-chain reads both come back empty —
|
||||
// there is no handle a program could re-enter run_code through.
|
||||
runCode: String(functions[RUN_CODE_NAME]),
|
||||
}),
|
||||
})
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(false)
|
||||
expect(JSON.parse((result.content[0] as { text: string }).text)).toEqual({ names: ['echo'], runCode: 'undefined' })
|
||||
})
|
||||
|
||||
it('renders byte-identical SDK text across consecutive assemblies of an unchanged tool set', async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'code' })
|
||||
registerEcho(ctx)
|
||||
const first = await systemPrompt.assemble()
|
||||
const second = await systemPrompt.assemble()
|
||||
const text = (assembly: typeof first) => assembly.sections.find(section => section.name === 'tools:sdk')?.text
|
||||
expect(text(first)).toBe(text(second))
|
||||
})
|
||||
|
||||
it('rejects every assembly when a non-native mode has no code runtime', async () => {
|
||||
const { systemPrompt } = await setup({ mode: 'code', runtime: false })
|
||||
await expect(systemPrompt.assemble()).rejects.toThrow(/requires a code runtime/)
|
||||
})
|
||||
|
||||
it("rejects every assembly when the runtime's language is not typescript", async () => {
|
||||
const { systemPrompt } = await setup({ mode: 'code', runtime: { language: 'python' } })
|
||||
await expect(systemPrompt.assemble()).rejects.toThrow(/language is "python"/)
|
||||
})
|
||||
|
||||
it("rejects the assembly when toolOrder names a native tool that mode 'code' no longer contributes", async () => {
|
||||
const { ctx, systemPrompt } = await setup({ mode: 'code', toolOrder: ['echo', '<unlisted-tools>'] })
|
||||
registerEcho(ctx)
|
||||
await expect(systemPrompt.assemble()).rejects.toThrow(/toolOrder lists unregistered tool "echo"/)
|
||||
})
|
||||
|
||||
it('removes run_code and the SDK section when the registry fiber disposes (HMR safety)', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt, {})
|
||||
await ctx.plugin(FakeRuntime, {})
|
||||
const fiber = await ctx.plugin(ToolRegistry, { mode: 'code' })
|
||||
expect(ctx.tools.get(RUN_CODE_NAME)).toBeDefined()
|
||||
await fiber.dispose()
|
||||
const assembly = await ctx.systemPrompt.assemble()
|
||||
expect(assembly.tools).toEqual([])
|
||||
expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('the run_code dispatch bridge', () => {
|
||||
it('bridges tool calls, returns only the curated output, and logs one event per dispatch', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
const { agent, events } = fakeAgent()
|
||||
runtime.behavior = async (request) => {
|
||||
const tools = request.bindings[0]!.functions
|
||||
const first = await tools.echo!({ value: 'one' })
|
||||
const second = await tools.echo!({ value: 'two' })
|
||||
return { logs: [{ source: 'console', level: 'log', text: `saw ${String(first)}` }], value: second }
|
||||
}
|
||||
const result = await runCode(ctx, 'const …: string = …', { agent })
|
||||
expect(result.isError).toBe(false)
|
||||
expect(result.content).toEqual([{ type: 'text', text: 'saw echo:one\necho:two' }])
|
||||
expect(calls).toEqual([{ value: 'one' }, { value: 'two' }])
|
||||
const dispatches = events.filter(event => event.type === 'tool/code-dispatch')
|
||||
expect(dispatches.map(event => event.data)).toEqual([
|
||||
{ parentCallId: 'call-1', subCallId: 'call-1:code:1', name: 'echo', arguments: { value: 'one' }, isError: false, resultSummary: 'echo:one' },
|
||||
{ parentCallId: 'call-1', subCallId: 'call-1:code:2', name: 'echo', arguments: { value: 'two' }, isError: false, resultSummary: 'echo:two' },
|
||||
])
|
||||
expect(result.meta).toEqual({ logs: [{ source: 'console', level: 'log', text: 'saw echo:one' }], dispatches: 2 })
|
||||
})
|
||||
|
||||
it('serializes Promise.all dispatches: tool executions never overlap, in submission order', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const intervals: [string, string][] = []
|
||||
let active = 0
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'probe',
|
||||
description: 'Records execution overlap.',
|
||||
parameters: { id: { type: 'string', required: true } },
|
||||
async execute(args) {
|
||||
active++
|
||||
expect(active, 'probe executions overlapped').toBe(1)
|
||||
intervals.push(['enter', args.id])
|
||||
await new Promise(resolve => setTimeout(resolve, 20))
|
||||
intervals.push(['exit', args.id])
|
||||
active--
|
||||
return [{ type: 'text' as const, text: args.id }]
|
||||
},
|
||||
}))
|
||||
runtime.behavior = async (request) => {
|
||||
const tools = request.bindings[0]!.functions
|
||||
const values = await Promise.all([tools.probe!({ id: 'a' }), tools.probe!({ id: 'b' }), tools.probe!({ id: 'c' })])
|
||||
return { logs: [], value: values.join(',') }
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(false)
|
||||
expect(intervals).toEqual([
|
||||
['enter', 'a'], ['exit', 'a'],
|
||||
['enter', 'b'], ['exit', 'b'],
|
||||
['enter', 'c'], ['exit', 'c'],
|
||||
])
|
||||
expect(result.content[0]).toEqual({ type: 'text', text: 'a,b,c' })
|
||||
})
|
||||
|
||||
it('rejects the program-side call when the tool errors, with the tool error text', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'fail',
|
||||
description: 'Always fails.',
|
||||
parameters: {},
|
||||
execute(): Promise<never> { return Promise.reject(new Error('deliberate failure')) },
|
||||
}))
|
||||
runtime.behavior = async (request) => {
|
||||
try {
|
||||
await request.bindings[0]!.functions.fail!({})
|
||||
return { logs: [], value: 'unreachable' }
|
||||
} catch (error: unknown) {
|
||||
return { logs: [], value: `caught: ${error instanceof Error ? error.message : String(error)}` }
|
||||
}
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.content[0]).toEqual({ type: 'text', text: 'caught: Error: deliberate failure' })
|
||||
})
|
||||
|
||||
it('a tools/pre-execute deny reaches the program as a binding rejection', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
registerEcho(ctx)
|
||||
ctx.on('tools/pre-execute', (exec, next) => {
|
||||
if (exec.name === 'echo') return Promise.resolve({ kind: 'deny' as const, reason: 'not on my watch' })
|
||||
return next()
|
||||
})
|
||||
runtime.behavior = async (request) => {
|
||||
try {
|
||||
await request.bindings[0]!.functions.echo!({ value: 'x' })
|
||||
return { logs: [], value: 'unreachable' }
|
||||
} catch (error: unknown) {
|
||||
return { logs: [], value: `denied: ${error instanceof Error ? error.message : String(error)}` }
|
||||
}
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.content[0]?.type).toBe('text')
|
||||
expect((result.content[0] as { text: string }).text).toContain('not on my watch')
|
||||
})
|
||||
|
||||
it('rejects a binding argument that does not survive JSON normalization, dispatching nothing', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
const { agent, events } = fakeAgent()
|
||||
runtime.behavior = async (request) => {
|
||||
try {
|
||||
await request.bindings[0]!.functions.echo!({ value: 'x', big: 1n })
|
||||
return { logs: [], value: 'unreachable' }
|
||||
} catch (error: unknown) {
|
||||
return { logs: [], value: error instanceof Error ? error.message : String(error) }
|
||||
}
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { agent })
|
||||
expect((result.content[0] as { text: string }).text).toContain('JSON-serializable')
|
||||
expect(calls).toEqual([])
|
||||
expect(events.filter(event => event.type === 'tool/code-dispatch')).toEqual([])
|
||||
})
|
||||
|
||||
it('dispatches the JSON-normalized value: what the tool sees is what the event logs', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
const { agent, events } = fakeAgent()
|
||||
runtime.behavior = async (request) => {
|
||||
// A Date survives structured clone but is not JSON; the bridge
|
||||
// normalizes it to its JSON form (an ISO string) BEFORE dispatch.
|
||||
await request.bindings[0]!.functions.echo!({ value: 'x', when: new Date(0) }).catch(() => undefined)
|
||||
return { logs: [] }
|
||||
}
|
||||
await runCode(ctx, 'program', { agent })
|
||||
expect(calls).toEqual([{ value: 'x', when: '1970-01-01T00:00:00.000Z' }])
|
||||
const dispatch = events.find(event => event.type === 'tool/code-dispatch')?.data as SessionEventMap['tool/code-dispatch']
|
||||
expect(dispatch.arguments).toEqual({ value: 'x', when: '1970-01-01T00:00:00.000Z' })
|
||||
})
|
||||
|
||||
it('suppresses sub-call additionalContext (deliberately; pinned)', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
registerEcho(ctx)
|
||||
ctx.on('tools/post-execute', (exec, _result, next): Promise<PostToolDecision> => {
|
||||
if (exec.name === 'echo') {
|
||||
return Promise.resolve({
|
||||
kind: 'accept' as const,
|
||||
additionalContext: { content: [{ type: 'text' as const, text: 'context for the next request' }], source: { kind: 'plugin' as const, plugin: 'test' } },
|
||||
})
|
||||
}
|
||||
return next()
|
||||
})
|
||||
runtime.behavior = async (request) => {
|
||||
await request.bindings[0]!.functions.echo!({ value: 'x' })
|
||||
return { logs: [], value: 'done' }
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(false)
|
||||
// The sub-call's context has no safe outlet mid-run; the parent result
|
||||
// must not carry it either.
|
||||
expect(result.additionalContext).toBeUndefined()
|
||||
})
|
||||
|
||||
it('converts a failed run into a structured isError result carrying kind, message, and captured logs', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
runtime.behavior = () => Promise.resolve({
|
||||
logs: [{ source: 'console', level: 'log', text: 'got this far' }],
|
||||
error: { kind: 'timeout', message: 'compute budget exhausted (300ms busy)' },
|
||||
})
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(true)
|
||||
expect(result.error).toEqual({ name: 'CodeRunFailedError', code: 'CODE_RUN_FAILED' })
|
||||
const text = (result.content[0] as { text: string }).text
|
||||
expect(text).toContain('code run failed (timeout)')
|
||||
expect(text).toContain('compute budget exhausted')
|
||||
expect(text).toContain('got this far')
|
||||
})
|
||||
|
||||
it('CodeRunFailedError is a HarnessError with the CODE_RUN_FAILED code', () => {
|
||||
const error = new CodeRunFailedError('boom')
|
||||
expect(error.code).toBe('CODE_RUN_FAILED')
|
||||
expect(error.name).toBe('CodeRunFailedError')
|
||||
})
|
||||
|
||||
it('aborting the outer signal aborts the in-flight sub-dispatch and abandons queued ones', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const seen: string[] = []
|
||||
let sawAbort = false
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'slow',
|
||||
description: 'Slow tool observing its signal.',
|
||||
parameters: { id: { type: 'string', required: true } },
|
||||
async execute(args, exec) {
|
||||
seen.push(args.id)
|
||||
await new Promise<void>((resolve) => {
|
||||
const timer = setTimeout(resolve, 500)
|
||||
exec.signal?.addEventListener('abort', () => { sawAbort = true; clearTimeout(timer); resolve() }, { once: true })
|
||||
})
|
||||
return [{ type: 'text' as const, text: args.id }]
|
||||
},
|
||||
}))
|
||||
const controller = new AbortController()
|
||||
runtime.behavior = async (request) => {
|
||||
const tools = request.bindings[0]!.functions
|
||||
const calls = [tools.slow!({ id: 'first' }).catch(() => 'rejected'), tools.slow!({ id: 'second' }).catch(() => 'rejected')]
|
||||
setTimeout(() => { controller.abort('user-cancel') }, 50)
|
||||
await Promise.all(calls)
|
||||
// A real runtime would be terminated by the abort; the fake honors the
|
||||
// contract by reporting the abort as the run failure.
|
||||
return { logs: [], error: { kind: 'abort', message: 'user-cancel' } }
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { signal: controller.signal })
|
||||
expect(result.isError).toBe(true)
|
||||
expect((result.content[0] as { text: string }).text).toContain('code run failed (abort)')
|
||||
expect(seen).toEqual(['first'])
|
||||
expect(sawAbort).toBe(true)
|
||||
})
|
||||
|
||||
it('a runtime that starts a binding call and then REJECTS still reaches quiescence before returning', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const { agent, events } = fakeAgent()
|
||||
let sawAbort = false
|
||||
let started!: () => void
|
||||
const inFlight = new Promise<void>((resolve) => { started = resolve })
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'slow',
|
||||
description: 'Slow tool observing its signal.',
|
||||
parameters: { id: { type: 'string', required: true } },
|
||||
async execute(args, exec) {
|
||||
started()
|
||||
await new Promise<void>((resolve) => {
|
||||
const timer = setTimeout(resolve, 500)
|
||||
exec.signal?.addEventListener('abort', () => { sawAbort = true; clearTimeout(timer); resolve() }, { once: true })
|
||||
})
|
||||
return [{ type: 'text' as const, text: args.id }]
|
||||
},
|
||||
}))
|
||||
runtime.behavior = async (request) => {
|
||||
// Start a sub-dispatch, keep its rejection held, and fail the run once
|
||||
// the tool is genuinely in flight — a seam error AFTER work has begun.
|
||||
// The bridge's settlement still owes quiescence: without the finally,
|
||||
// run_code would return now and the slow tool would finish (and log)
|
||||
// afterwards.
|
||||
request.bindings[0]!.functions.slow!({ id: 'orphan' }).catch(() => 'held')
|
||||
await inFlight
|
||||
throw new Error('backend exploded')
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { agent })
|
||||
expect(result.isError).toBe(true)
|
||||
expect((result.content[0] as { text: string }).text).toContain('backend exploded')
|
||||
// Quiescence held: the in-flight sub-dispatch was aborted and its event
|
||||
// logged INSIDE the run_code execution, not after it returned.
|
||||
expect(sawAbort).toBe(true)
|
||||
expect(events.filter(event => event.type === 'tool/code-dispatch').map(event => (event.data as { name: string }).name)).toEqual(['slow'])
|
||||
})
|
||||
|
||||
it('runs without an owning agent: dispatches work, event logging is skipped', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
runtime.behavior = async (request) => {
|
||||
await request.bindings[0]!.functions.echo!({ value: 'x' })
|
||||
return { logs: [], value: 'ok' }
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(false)
|
||||
expect(calls).toEqual([{ value: 'x' }])
|
||||
})
|
||||
|
||||
it('executing run_code under a missing runtime is a structured isError, not a crash', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt, {})
|
||||
await ctx.plugin(ToolRegistry, { mode: 'code' })
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(true)
|
||||
expect((result.content[0] as { text: string }).text).toContain('requires a code runtime')
|
||||
})
|
||||
|
||||
it('presents the PROGRAM as the execute-card title on both call and result (the one slot execute cards always show)', async () => {
|
||||
const { ctx } = await setup({ mode: 'code' })
|
||||
const tool = ctx.tools.get(RUN_CODE_NAME)!
|
||||
// The program IS the title, mirroring how command tools title their cards
|
||||
// with the command: an ACP client's execute-card header is the only
|
||||
// always-visible slot (Zed renders no body content and no raw input for
|
||||
// execute-kind cards without a real terminal).
|
||||
expect(tool.presentCall?.({ code: 'return 1' })).toEqual({
|
||||
card: 'generic',
|
||||
title: 'return 1',
|
||||
kind: 'execute',
|
||||
rawInput: 'return 1',
|
||||
})
|
||||
const view = tool.presentResult?.({ code: 'return 1' }, {
|
||||
content: [{ type: 'text', text: 'model-facing' }],
|
||||
isError: false,
|
||||
meta: { logs: [{ source: 'console', level: 'log', text: 'printed' }], dispatches: 1 },
|
||||
})
|
||||
// The result omits the title — an update replaces only provided fields,
|
||||
// so the pending card's program title persists through completion.
|
||||
expect(view).toEqual({
|
||||
card: 'generic',
|
||||
content: [{ type: 'text', text: 'printed' }],
|
||||
})
|
||||
// No captured output → no content either; everything pending persists.
|
||||
expect(tool.presentResult?.({ code: 'x' }, { content: [], isError: false, meta: { logs: [], dispatches: 2 } }))
|
||||
.toEqual({ card: 'generic' })
|
||||
// Replay with an unrecognizable meta falls back to the generic rendering.
|
||||
expect(tool.presentResult?.({ code: 'x' }, { content: [], isError: false, meta: { other: true } })).toBeUndefined()
|
||||
expect(tool.presentResult?.({ code: 'x' }, { content: [], isError: false })).toBeUndefined()
|
||||
})
|
||||
|
||||
it('renders non-text sub-result blocks as placeholders and truncates long event summaries', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const { agent, events } = fakeAgent()
|
||||
const long = 'x'.repeat(300)
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'mixed',
|
||||
description: 'Returns mixed content.',
|
||||
parameters: {},
|
||||
execute() {
|
||||
return Promise.resolve([
|
||||
{ type: 'text' as const, text: long },
|
||||
{ type: 'reasoning' as const, text: 'hidden' },
|
||||
])
|
||||
},
|
||||
}))
|
||||
runtime.behavior = async (request) => {
|
||||
const value = await request.bindings[0]!.functions.mixed!({})
|
||||
return { logs: [], value }
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { agent })
|
||||
expect(result.isError).toBe(false)
|
||||
expect((result.content[0] as { text: string }).text).toBe(`${long}\n[reasoning content]`)
|
||||
const dispatch = events.find(event => event.type === 'tool/code-dispatch')?.data as SessionEventMap['tool/code-dispatch']
|
||||
expect(dispatch.resultSummary.length).toBe(201)
|
||||
expect(dispatch.resultSummary.endsWith('…')).toBe(true)
|
||||
})
|
||||
|
||||
it('rejects undefined, JSON-throwing, and JSON-unrepresentable binding arguments BEFORE dispatch', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
const { agent, events } = fakeAgent()
|
||||
runtime.behavior = async (request) => {
|
||||
const echo = request.bindings[0]!.functions.echo!
|
||||
const catchMessage = (promise: Promise<unknown>) => promise.then(() => 'resolved', (error: unknown) => error instanceof Error ? error.message : String(error))
|
||||
return {
|
||||
logs: [],
|
||||
value: [
|
||||
// Root undefined must reject up front: the event log rejects it as
|
||||
// data, and nothing may execute unlogged.
|
||||
await catchMessage(echo(undefined)),
|
||||
// A toJSON that throws a NON-Error propagates out of JSON.stringify.
|
||||
await catchMessage(echo({ toJSON() { throw 'raw-throw' } })),
|
||||
// A bare function is a value JSON cannot represent at all.
|
||||
await catchMessage(echo(() => 1)),
|
||||
].join(' | '),
|
||||
}
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { agent })
|
||||
const text = (result.content[0] as { text: string }).text
|
||||
expect(text).toContain('call the tool with an arguments object')
|
||||
expect(text).toContain('JSON-serializable: raw-throw')
|
||||
expect(text).toContain('a value JSON cannot represent')
|
||||
// None of the three dispatched, none logged.
|
||||
expect(calls).toEqual([])
|
||||
expect(events.filter(event => event.type === 'tool/code-dispatch')).toEqual([])
|
||||
})
|
||||
|
||||
it('logs the value the tool RECEIVED even when the tool mutates its arguments', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const { agent, events } = fakeAgent()
|
||||
ctx.tools.register(defineTool({
|
||||
name: 'mutator',
|
||||
description: 'Mutates its own args object.',
|
||||
parameters: { list: { type: 'array', required: true } },
|
||||
execute(args) {
|
||||
args.list.push('injected-by-tool')
|
||||
return Promise.resolve([{ type: 'text' as const, text: 'mutated' }])
|
||||
},
|
||||
}))
|
||||
runtime.behavior = async (request) => {
|
||||
await request.bindings[0]!.functions.mutator!({ list: ['original'] })
|
||||
return { logs: [] }
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { agent })
|
||||
expect(result.isError).toBe(false)
|
||||
const dispatch = events.find(event => event.type === 'tool/code-dispatch')?.data as SessionEventMap['tool/code-dispatch']
|
||||
expect(dispatch.arguments).toEqual({ list: ['original'] })
|
||||
})
|
||||
|
||||
it('exposes a tool named __proto__ as an ordinary own binding', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
ctx.tools.register(defineTool({
|
||||
name: '__proto__',
|
||||
description: 'A prototype-colliding tool name.',
|
||||
parameters: {},
|
||||
execute() { return Promise.resolve([{ type: 'text' as const, text: 'proto-tool-ok' }]) },
|
||||
}))
|
||||
runtime.behavior = async (request) => {
|
||||
const functions = request.bindings[0]!.functions
|
||||
expect(Object.getPrototypeOf(functions)).toBeNull()
|
||||
const value = await functions['__proto__']!({})
|
||||
return { logs: [], value }
|
||||
}
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect(result.isError).toBe(false)
|
||||
expect(result.content[0]).toEqual({ type: 'text', text: 'proto-tool-ok' })
|
||||
})
|
||||
|
||||
it('renders a non-string completion value inspect-style', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
runtime.behavior = () => Promise.resolve({ logs: [], value: { n: 42 } })
|
||||
const result = await runCode(ctx, 'program')
|
||||
expect((result.content[0] as { text: string }).text).toBe('{ n: 42 }')
|
||||
})
|
||||
|
||||
it('reports a pre-aborted outer signal as the run failure without dispatching anything', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
runtime.behavior = (request) => {
|
||||
// The fake honors the seam contract for an already-aborted signal.
|
||||
if (request.signal?.aborted) return Promise.resolve({ logs: [], error: { kind: 'abort' as const, message: String(request.signal.reason) } })
|
||||
return Promise.resolve({ logs: [], value: 'unreachable' })
|
||||
}
|
||||
const controller = new AbortController()
|
||||
controller.abort('too-late')
|
||||
const result = await runCode(ctx, 'program', { signal: controller.signal })
|
||||
expect(result.isError).toBe(true)
|
||||
expect((result.content[0] as { text: string }).text).toContain('code run failed (abort)')
|
||||
expect(calls).toEqual([])
|
||||
})
|
||||
|
||||
it('rejects a binding invoked after the run is over without dispatching it', async () => {
|
||||
const { ctx, runtime } = await setup({ mode: 'code' })
|
||||
const calls = registerEcho(ctx)
|
||||
const controller = new AbortController()
|
||||
runtime.behavior = async (request) => {
|
||||
controller.abort('cancelled-mid-run')
|
||||
const message = await request.bindings[0]!.functions.echo!({ value: 'x' })
|
||||
.then(() => 'resolved', (error: unknown) => error instanceof Error ? error.message : String(error))
|
||||
return { logs: [], value: message }
|
||||
}
|
||||
const result = await runCode(ctx, 'program', { signal: controller.signal })
|
||||
expect(result.isError).toBe(false)
|
||||
expect((result.content[0] as { text: string }).text).toContain('not dispatched')
|
||||
expect(calls).toEqual([])
|
||||
})
|
||||
|
||||
it('a tool/code-dispatch event never derives a model message', () => {
|
||||
const session = new Session(SessionId('code-mode-derive'))
|
||||
session.append('user/message', { content: [{ type: 'text', text: 'hi' }], source: { kind: 'user' } }, { surfaceOp: 'append' })
|
||||
session.append('tool/code-dispatch', {
|
||||
parentCallId: CallId('p1'),
|
||||
subCallId: CallId('p1:code:1'),
|
||||
name: 'echo',
|
||||
arguments: { value: 'x' },
|
||||
isError: false,
|
||||
resultSummary: 'echo:x',
|
||||
})
|
||||
const derived = session.deriveMessages()
|
||||
expect(derived).toHaveLength(1)
|
||||
expect(derived[0]?.role).toBe('user')
|
||||
})
|
||||
|
||||
it('defaults to native mode under direct construction with no config', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt, {})
|
||||
const registry = new ToolRegistry(ctx)
|
||||
expect(registry.get(RUN_CODE_NAME)).toBeUndefined()
|
||||
const assembly = await ctx.systemPrompt.assemble()
|
||||
expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
|
||||
})
|
||||
})
|
||||
@@ -35,7 +35,7 @@ describe('gen-tool-catalog collectToolCatalog', () => {
|
||||
it('boots every shipped tool package and harvests its model-facing schemas', async () => {
|
||||
const catalog = await collectToolCatalog()
|
||||
const names = catalog.flatMap(entry => entry.schemas.map(s => s.name)).sort()
|
||||
expect(names).toEqual(['ask_user_question', 'bash', 'bash_kill', 'bash_output', 'cordis_inspect', 'cordis_mount', 'cordis_unmount', 'edit', 'read', 'skill', 'subagent', 'todo_write', 'web_fetch', 'web_search', 'write'])
|
||||
expect(names).toEqual(['ask_user_question', 'bash', 'bash_kill', 'bash_output', 'cordis_inspect', 'cordis_mount', 'cordis_unmount', 'edit', 'read', 'run_code', 'skill', 'subagent', 'todo_write', 'web_fetch', 'web_search', 'workflow', 'write'])
|
||||
// Every tool carries a JSON-Schema `parameters` object (what the model sees).
|
||||
for (const entry of catalog) {
|
||||
for (const schema of entry.schemas) {
|
||||
|
||||
124
packages/core/tools/tests/ts-types.spec.ts
Normal file
124
packages/core/tools/tests/ts-types.spec.ts
Normal file
@@ -0,0 +1,124 @@
|
||||
import { describe, expect, it } from 'vitest'
|
||||
import { jsonSchemaToTs, renderToolsSdk } from '@deepseek-ai/dsh-tools/src/ts-types.ts'
|
||||
import { schemaSpecToJsonSchema } from '@deepseek-ai/dsh-tools'
|
||||
import type { ToolSchema } from '@deepseek-ai/dsh-llm'
|
||||
|
||||
describe('jsonSchemaToTs', () => {
|
||||
it('maps the defineTool DSL subset', () => {
|
||||
const cases: [unknown, string][] = [
|
||||
[{ type: 'string' }, 'string'],
|
||||
[{ type: 'number' }, 'number'],
|
||||
[{ type: 'boolean' }, 'boolean'],
|
||||
[{ type: 'string', enum: ['a', 'b'] }, '"a" | "b"'],
|
||||
[{ type: 'array', items: { type: 'number' } }, 'number[]'],
|
||||
[{ type: 'array', items: { type: 'string', enum: ['x', 'y'] } }, '("x" | "y")[]'],
|
||||
[{ type: 'array' }, 'unknown[]'],
|
||||
[{ type: 'object' }, 'Record<string, unknown>'],
|
||||
[{ type: 'object', properties: {} }, 'Record<string, unknown>'],
|
||||
]
|
||||
for (const [schema, expected] of cases) {
|
||||
expect(jsonSchemaToTs(schema), JSON.stringify(schema)).toBe(expected)
|
||||
}
|
||||
})
|
||||
|
||||
it('renders objects with required/optional keys, nested shapes, and per-property docs', () => {
|
||||
const schema = schemaSpecToJsonSchema({
|
||||
path: { type: 'string', required: true, description: 'Absolute file path' },
|
||||
limit: { type: 'number' },
|
||||
opts: {
|
||||
type: 'object',
|
||||
properties: { deep: { type: 'boolean', required: true } },
|
||||
},
|
||||
})
|
||||
expect(jsonSchemaToTs(schema)).toBe([
|
||||
'{',
|
||||
' /** Absolute file path */',
|
||||
' path: string;',
|
||||
' limit?: number;',
|
||||
' opts?: {',
|
||||
' deep: boolean;',
|
||||
' };',
|
||||
'}',
|
||||
].join('\n'))
|
||||
})
|
||||
|
||||
it('is total: unsupported or hostile constructs degrade to unknown, never throw', () => {
|
||||
const cases: unknown[] = [
|
||||
undefined,
|
||||
null,
|
||||
42,
|
||||
'string-schema',
|
||||
{},
|
||||
{ type: 'integer' },
|
||||
{ type: 'null' },
|
||||
{ oneOf: [{ type: 'string' }] },
|
||||
{ $ref: '#/defs/x' },
|
||||
{ type: 'object', properties: 7 },
|
||||
{ type: 'object', properties: { bad: { $ref: 'x' } } },
|
||||
{ type: 'string', enum: [1, 2] },
|
||||
{ type: 'string', enum: [] },
|
||||
]
|
||||
for (const schema of cases) {
|
||||
expect(() => jsonSchemaToTs(schema), JSON.stringify(schema)).not.toThrow()
|
||||
}
|
||||
expect(jsonSchemaToTs({ type: 'integer' })).toBe('unknown')
|
||||
expect(jsonSchemaToTs({ oneOf: [] })).toBe('unknown')
|
||||
expect(jsonSchemaToTs({ type: 'object', properties: 7 })).toBe('Record<string, unknown>')
|
||||
expect(jsonSchemaToTs({ type: 'object', properties: { bad: { $ref: 'x' } }, required: ['bad'] })).toContain('bad: unknown;')
|
||||
// A non-string-only enum degrades to plain string; an empty one too.
|
||||
expect(jsonSchemaToTs({ type: 'string', enum: [1, 2] })).toBe('string')
|
||||
expect(jsonSchemaToTs({ type: 'string', enum: [] })).toBe('string')
|
||||
// A hostile required list only accepts string members.
|
||||
expect(jsonSchemaToTs({ type: 'object', properties: { a: { type: 'string' } }, required: [7] })).toContain('a?: string;')
|
||||
// A property VALUE that is not an object degrades to unknown (and can
|
||||
// carry no description).
|
||||
expect(jsonSchemaToTs({ type: 'object', properties: { weird: 42 } })).toContain('weird?: unknown;')
|
||||
})
|
||||
|
||||
it('escapes a comment-closer inside a description so the generated JSDoc cannot end early', () => {
|
||||
const rendered = jsonSchemaToTs({
|
||||
type: 'object',
|
||||
properties: { glob: { type: 'string', description: 'a pattern like packages/*/tool-*/ over here' } },
|
||||
})
|
||||
expect(rendered).not.toContain('tool-*/ over')
|
||||
expect(rendered).toContain(String.raw`tool-*\/ over`)
|
||||
})
|
||||
})
|
||||
|
||||
describe('renderToolsSdk', () => {
|
||||
const bash: ToolSchema = {
|
||||
name: 'bash',
|
||||
description: 'Run a shell command.',
|
||||
parameters: schemaSpecToJsonSchema({ command: { type: 'string', required: true } }) as unknown as Record<string, unknown>,
|
||||
}
|
||||
const exotic: ToolSchema = {
|
||||
name: 'my-mcp.tool',
|
||||
description: 'Exotic name.',
|
||||
parameters: schemaSpecToJsonSchema({}) as unknown as Record<string, unknown>,
|
||||
}
|
||||
|
||||
it('declares every tool in lexicographic order with quoted keys for exotic names', () => {
|
||||
const text = renderToolsSdk([exotic, bash])
|
||||
expect(text).toContain('declare const tools: {')
|
||||
expect(text.indexOf('bash(args:')).toBeGreaterThan(0)
|
||||
expect(text).toContain('"my-mcp.tool"(args:')
|
||||
expect(text.indexOf('bash(args:')).toBeLessThan(text.indexOf('"my-mcp.tool"(args:'))
|
||||
expect(text).toContain('): Promise<string>;')
|
||||
expect(text).toContain('/** Run a shell command. */')
|
||||
// The fixed instruction lines the model relies on.
|
||||
expect(text).toContain('erasable syntax only')
|
||||
expect(text).toContain('rejects with an `Error`')
|
||||
expect(text).toContain('sequentially, even under `Promise.all`')
|
||||
expect(text).toContain('JSON-serializable')
|
||||
})
|
||||
|
||||
it('is deterministic: same tool set, byte-identical text regardless of input order', () => {
|
||||
expect(renderToolsSdk([bash, exotic])).toBe(renderToolsSdk([exotic, bash]))
|
||||
// Equal names sort stably (the comparator's equal arm).
|
||||
expect(renderToolsSdk([bash, bash])).toBe(renderToolsSdk([bash, bash]))
|
||||
})
|
||||
|
||||
it('renders an empty declaration for an empty tool set', () => {
|
||||
expect(renderToolsSdk([])).toContain('declare const tools: {}')
|
||||
})
|
||||
})
|
||||
@@ -8,6 +8,12 @@
|
||||
"src"
|
||||
],
|
||||
"references": [
|
||||
{
|
||||
"path": "../../core/session"
|
||||
},
|
||||
{
|
||||
"path": "../../code-runtime/code-runtime"
|
||||
},
|
||||
{
|
||||
"path": "../../../vendor/cosmokit"
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user