mirror of
https://github.com/deepseek-ai/deepseek-harness
synced 2026-08-15 21:04:50 +00:00
# Conflicts: # docs/config-catalog.md # docs/module-graph.md # docs/rfc/INDEX.md # examples/acp-agent/composition.md # examples/acp-agent/cordis.yml # examples/acp-agent/tests/snapshots/advanced-toolchain/session.1.jsonl # examples/acp-agent/tests/snapshots/advanced-toolchain/session.2.jsonl # examples/acp-agent/tests/snapshots/advanced-toolchain/session.jsonl # examples/acp-agent/tests/snapshots/both-mode-turn/session.jsonl # examples/acp-agent/tests/snapshots/permission-switching/session.jsonl # examples/acp-agent/tests/snapshots/skill-load/session.jsonl # examples/acp-agent/tests/snapshots/text-turn/session.jsonl # packages/bash/bash-sandbox/package.json # packages/bash/bash/package.json # packages/bash/tool-bash/src/index.ts # packages/fs/fs/package.json # packages/fs/tool-fs/package.json # packages/fs/tool-fs/src/index.ts # packages/fs/tool-fs/tests/tools.spec.ts # pnpm-lock.yaml
534 lines
29 KiB
TypeScript
534 lines
29 KiB
TypeScript
/**
|
|
* The model-facing bash tools: `bash`, `bash_output`, `bash_kill`. Pure
|
|
* schema + text shaping — every process concern lives behind the `ctx.bash`
|
|
* executor seam (`@deepseek-ai/dsh-bash`), so sandbox/permission/remote
|
|
* executor implementations swap in without touching what the model sees.
|
|
*
|
|
* Background notifications: when a background task completes, a short notice
|
|
* is injected into the owning agent's session (`agent.inject()` — the
|
|
* documented context seam). Injection is durable context for the NEXT model
|
|
* request, not a wake-up: an idle agent stays idle until something sends a
|
|
* message, which is why the tool descriptions tell the model to poll with
|
|
* `bash_output`.
|
|
*
|
|
* Task ownership: a background task's OWNER is an opaque token — the owning
|
|
* agent's `session.header.id` — passed to the executor at spawn
|
|
* (`resolve({ …, owner })`) and stored ON THE TASK inside the executor
|
|
* (`@deepseek-ai/dsh-bash`'s `ownerOf(id)` seam), NOT in a plugin-local map.
|
|
* `bash_output`/`bash_kill` compare `ctx.bash.ownerOf(id)` to the caller's token
|
|
* and reject a task owned by a DIFFERENT session (`owner !== undefined && owner
|
|
* !== caller`); an unowned task (no token — started by a non-agent caller) is
|
|
* open to anyone. Task ids are global and predictable (`bash-1`, …); under
|
|
* multi-session ACP (RFC 011) this token check is the fence that stops one
|
|
* session's agent from reading or killing another session's background task.
|
|
*
|
|
* Storing the token on the task in the EXECUTOR (disposed with the `dsh-bash`
|
|
* fiber), rather than in this plugin, is what makes ownership survive a
|
|
* `tool-bash` HMR reload — a reload that reset a plugin-local map would orphan
|
|
* a task spawned before it. (The `onTaskDone` listener is still effect-scoped
|
|
* to this plugin's `apply`, so a
|
|
* completion landing during the reload gap still drops its one notice — the
|
|
* pre-existing reload-gap drop — but the ownership fence itself is HMR-proof.)
|
|
*
|
|
* Commands run with the executor's full authority unless a sandboxing
|
|
* executor (`@deepseek-ai/dsh-bash-sandbox`) confines them; per-call
|
|
* allow/deny/ask policy is the `tools/pre-execute` waterfall — see
|
|
* docs/architecture.md § Extension And Composition. Under a sandboxing
|
|
* executor this plugin also advertises the ESCALATION surface
|
|
* (`sandbox_permissions`/`justification` — the sandbox RFC § Escalation,
|
|
* docs/rfc/implemented/feature/2026-07-06-sandbox.md): a command the
|
|
* sandbox denied may be retried once under a strictly wider mode, resolved
|
|
* through `ctx.approval` BEFORE anything executes and failing closed on every
|
|
* unanswerable path. The fields exist only when the mounted executor reports
|
|
* a confining default (`ctx.bash.sandboxMode`) — a lever is never advertised
|
|
* that the composition cannot honor.
|
|
*
|
|
* Per-session mode switching (the sandbox RFC § Per-session mode switching): a session may carry a
|
|
* standing sandbox-mode override — the `sandbox/mode` event fold from
|
|
* `@deepseek-ai/dsh-bash` — which this plugin makes real at EXECUTION: each
|
|
* call is stamped `escalation grant > session override > executor default`.
|
|
* The prompt deliberately does NOT state the mode and no switch is narrated:
|
|
* the model learns the boundary from the denial marker (which names the mode
|
|
* it ran under) exactly when it matters, instead of preemptively refusing
|
|
* work a standing declaration would discourage.
|
|
*
|
|
* @module @deepseek-ai/dsh-tool-bash
|
|
*/
|
|
|
|
import type { Context } from 'cordis'
|
|
import { isAbsolute, resolve as resolvePath } from 'node:path'
|
|
import { defineTool } from '@deepseek-ai/dsh-tools'
|
|
import type { GenericCallView, TerminalCallView, ToolExecution, ToolResult, ToolResultView } from '@deepseek-ai/dsh-tools'
|
|
import type { Agent } from '@deepseek-ai/dsh-agent'
|
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
// Side-effect type import: declaration-merges `ctx.approval`, consumed
|
|
// opportunistically by the escalation gate (`ctx.get('approval')` — the seam
|
|
// stays optional at runtime, same pattern as dsh-tools' ask routing).
|
|
import type {} from '@deepseek-ai/dsh-user-approval'
|
|
import type { SandboxMode } from '@deepseek-ai/dsh-sandbox'
|
|
import {
|
|
ESCALATION_TARGETS,
|
|
approveEscalation,
|
|
escalationHintMarker,
|
|
sandboxDenialMarker,
|
|
validateEscalationArgs,
|
|
} from '@deepseek-ai/dsh-sandbox'
|
|
import { effectiveSandboxMode } from '@deepseek-ai/dsh-sandbox-policy'
|
|
import { BashTaskId, OwnerToken } from '@deepseek-ai/dsh-bash'
|
|
import type { BashTask } from '@deepseek-ai/dsh-bash'
|
|
import { parseExitStatus, renderResult } from './render.ts'
|
|
|
|
export const name = 'tool-bash'
|
|
export const inject = ['tools', 'bash', 'systemPrompt']
|
|
|
|
/**
|
|
* Validate the constraints the SchemaSpec can't express. `defineTool` now
|
|
* validates parsed args against the SchemaSpec before `execute` runs (the
|
|
* arg-validation RFC), so type/required/enum checks are already done and `args`
|
|
* is the validated `InferArgs` shape here. What remains are value constraints
|
|
* the DSL has no vocabulary for: non-empty strings, a positive finite timeout,
|
|
* and the escalation pairing (`sandbox_permissions` and `justification` travel
|
|
* together — an approval prompt without a reason, or a reason driving nothing,
|
|
* is a malformed ask).
|
|
*/
|
|
function validateBashArgs(args: BashToolArgs): void {
|
|
if (args.command.trim().length === 0) {
|
|
throw new Error('invalid command: expected a non-empty string')
|
|
}
|
|
if (args.description.trim().length === 0) {
|
|
throw new Error('invalid description: expected a non-empty string')
|
|
}
|
|
if (args.timeoutMs !== undefined && (!Number.isFinite(args.timeoutMs) || args.timeoutMs <= 0)) {
|
|
throw new Error(`invalid timeoutMs: expected a positive number, got ${JSON.stringify(args.timeoutMs)}`)
|
|
}
|
|
// The escalation pairing (sandbox_permissions ⇔ justification, non-empty) is
|
|
// the shared rule both enforcing families validate identically.
|
|
validateEscalationArgs(args.sandbox_permissions, args.justification)
|
|
}
|
|
|
|
/**
|
|
* Reject an empty `task_id`. Type and presence are guaranteed by the
|
|
* SchemaSpec validation (the arg-validation RFC); only the non-empty constraint, which the
|
|
* DSL can't express, is left to check here.
|
|
*/
|
|
function validateTaskId(value: string): BashTaskId {
|
|
if (value.length === 0) {
|
|
throw new Error(`invalid task_id: expected a string, got ${JSON.stringify(value)}`)
|
|
}
|
|
return BashTaskId(value)
|
|
}
|
|
|
|
/**
|
|
* The bash tool's validated argument shape — the base parameters plus the two
|
|
* escalation fields, which are ADVERTISED only when the mounted executor
|
|
* reports a confining default mode (absent from the schema otherwise, so the
|
|
* SchemaSpec validator rejects them before `execute` ever sees one).
|
|
*/
|
|
interface BashToolArgs {
|
|
command: string
|
|
description: string
|
|
timeoutMs?: number
|
|
workdir?: string
|
|
run_in_background?: boolean
|
|
sandbox_permissions?: string
|
|
justification?: string
|
|
}
|
|
|
|
/**
|
|
* The bash tool's static description. The base text is byte-stable regardless
|
|
* of composition (it is part of the pinned snapshot header); the escalation
|
|
* teaching rides only when the mounted executor actually honors the fields —
|
|
* it names the ONE sanctioned exception to the base text's "do not retry
|
|
* another way" rule. Its deference clause ("If the session states approval
|
|
* prompts are disabled…") points at the approval plugin's never-policy prompt
|
|
* sentence by meaning, not by parsed wording — a rendezvous kept working by
|
|
* that sentence continuing to open with the approvals-disabled claim.
|
|
*/
|
|
function bashDescription(escalationModes: readonly SandboxMode[]): string {
|
|
const base = 'Execute a bash command (`bash -c`) and return its stdout/stderr. '
|
|
+ 'Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — '
|
|
+ 'pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. '
|
|
+ 'Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way (a background task reports the same marker via bash_output once it has finished). '
|
|
+ 'Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. '
|
|
+ 'Set `run_in_background: true` for long-running commands: the call returns a task id immediately; '
|
|
+ 'poll it with `bash_output` and stop it with `bash_kill`.'
|
|
if (escalationModes.length === 0) return base
|
|
return base + ' Attempting a command the sandbox may deny is safe and expected: run it and read the '
|
|
+ 'marker rather than assuming the denial. When a command IS denied and a wider mode would let it '
|
|
+ 'succeed, escalate immediately in the SAME turn — the ONE sanctioned exception to a denial: retry '
|
|
+ 'the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) '
|
|
+ 'plus a one-sentence `justification`. Do not detour through chat to ask permission first — the '
|
|
+ 'approval prompt raised by that retry IS how the user consents. If the session states approval '
|
|
+ 'prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. '
|
|
+ 'Never escalate speculatively: ground the request in a real denial — normally the one THIS command '
|
|
+ 'just hit; escalating up front is fine only when this session already denied the same access. '
|
|
+ 'A rejected escalation is final for THAT command — stop and explain, never work around '
|
|
+ 'it — but it does not forbid attempting or escalating other commands later.'
|
|
}
|
|
|
|
// Pure tool-owned presentation used for both live events and replay.
|
|
|
|
/**
|
|
* Pending-state presentation for a `bash` call. The TITLE is the exact `command`
|
|
* — a `kind: 'execute'` card is rendered as a terminal whose header label IS the
|
|
* title, and an execute-kind card HIDES `rawInput` (Zed: `should_show_raw_input
|
|
* = !is_terminal_tool`), so the command must BE the title to be seen. This
|
|
* mirrors the reference ACP adapters (claude-agent-acp, codex-acp), which both
|
|
* use the bare command as an execute tool's title. The model-written
|
|
* `description` (a readable summary) rides as a `content` text block shown ABOVE
|
|
* the card. (Note: claude-agent-acp DROPS the description in terminal mode and
|
|
* shows only the card; surfacing it as a content block is a deliberate
|
|
* divergence here — we keep the human summary visible alongside the card.)
|
|
* `rawInput` still carries the bare command for non-execute UIs that DO render it.
|
|
*
|
|
* `terminal` marks the call so a capable UI renders a TERMINAL card — but ONLY a
|
|
* FOREGROUND run is a terminal: a `run_in_background` call returns a task id
|
|
* immediately (it never streams a terminal; its output is polled via
|
|
* `bash_output`), so it is NOT marked terminal and renders as an ordinary
|
|
* execute card. For a foreground run the `terminal.cwd` (header) is the model
|
|
* `workdir` when given — ABSOLUTE as-is, RELATIVE for the UI bridge to resolve
|
|
* against the session cwd; when omitted the bridge fills the session workspace
|
|
* cwd (this PURE presenter, args only, can't see it).
|
|
*/
|
|
type BashCallArgs = { command: string; description: string; workdir?: string; run_in_background?: boolean }
|
|
|
|
function presentBashCall(args: BashCallArgs): GenericCallView | TerminalCallView {
|
|
// A background start is not an interactive terminal — a generic execute card
|
|
// with the command as rawInput and the description as a content block.
|
|
if (args.run_in_background === true) {
|
|
return {
|
|
card: 'generic',
|
|
title: args.command,
|
|
kind: 'execute',
|
|
rawInput: args.command,
|
|
content: [{ type: 'text', text: args.description }],
|
|
}
|
|
}
|
|
// A foreground run IS a terminal: the command titles the card, the description
|
|
// renders above it, and the cwd (when the model gave a workdir) heads it.
|
|
return {
|
|
card: 'terminal',
|
|
title: args.command,
|
|
description: args.description,
|
|
...args.workdir !== undefined ? { cwd: args.workdir } : {},
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Completed-state presentation for a `bash` call. Two parallel renderings of the
|
|
* same output: `terminal.output` for a UI that shows a terminal card (the run's
|
|
* stdout/stderr + status markers, exactly as the model sees them — the RAW text,
|
|
* newlines preserved, since a terminal renderer relies on exact bytes), and a
|
|
* fenced ```console `content` block as the fallback for a UI without terminal
|
|
* support (the fences are a UI-only affordance, so they live here, not in the
|
|
* model-facing result; the fenced body is trimmed of trailing blank lines for a
|
|
* tidy block). A capable UI also gets an exit-status pill from `terminal.exitCode`
|
|
* / `terminal.signal`, parsed from the status markers `renderResult` appended.
|
|
*
|
|
* Terminal output/exit is suppressed for results that are NOT a finished
|
|
* foreground run: a `run_in_background` start (`isBackground` — the text is a
|
|
* task-id ack, not a streamed run) and an `isError` result (a spawn failure or
|
|
* abort — there is no real process exit to pill, and the body is an error
|
|
* message, not `renderResult` output, so parsing it would be meaningless). Those
|
|
* return a `generic` result whose content is the fenced ```console block. A
|
|
* finished foreground run returns a `terminal` result carrying the RAW output
|
|
* and the parsed exit status; the BRIDGE derives the fenced fallback from
|
|
* `output` for a UI without terminal support, so the tool does not double-encode
|
|
* it. A non-text result (unexpected for bash) falls through to `undefined`.
|
|
*/
|
|
function presentBashResult(args: unknown, result: ToolResult): ToolResultView | undefined {
|
|
const block = result.content.length === 1 ? result.content[0] : undefined
|
|
if (block === undefined || block.type !== 'text') return undefined
|
|
const raw = block.text
|
|
const isBackground = typeof args === 'object' && args !== null && (args as { run_in_background?: unknown }).run_in_background === true
|
|
// A background ack or an errored run is not a real terminal exit: render the
|
|
// fenced ```console fallback as generic content (no exit pill).
|
|
if (isBackground || result.isError) {
|
|
return { card: 'generic', content: [{ type: 'text', text: `\`\`\`console\n${raw.replace(/\n+$/, '')}\n\`\`\`` }] }
|
|
}
|
|
// A finished foreground run: RAW output + parsed exit for the terminal card.
|
|
// The bridge derives the no-capability fenced fallback from `output`.
|
|
return { card: 'terminal', output: raw, ...parseExitStatus(raw) }
|
|
}
|
|
|
|
/** Pending-state presentation for `bash_output`/`bash_kill` (background-task tools). */
|
|
function presentTaskCall(verb: string, args: { task_id: string }): GenericCallView {
|
|
return { card: 'generic', title: `${verb} background task ${args.task_id}`, kind: 'execute', rawInput: args.task_id }
|
|
}
|
|
|
|
/**
|
|
* Resolve the working directory for a bash call. Precedence: an explicit model
|
|
* `workdir` wins; otherwise default to the calling agent's session cwd
|
|
* (`session.header.cwd`) so each ACP session's commands run in ITS workspace,
|
|
* not the server's launch dir. A RELATIVE model `workdir` is resolved against
|
|
* the session cwd (the tool tells the model to pass `workdir` instead of `cd`,
|
|
* so a relative one should be relative to the session's root, not `process.cwd()`).
|
|
* Returns `undefined` when neither is available (no agent / headerless session /
|
|
* no session cwd) — the executor then applies its own config/`process.cwd()`
|
|
* default, preserving today's non-ACP behavior.
|
|
*/
|
|
function resolveWorkdir(modelWorkdir: string | undefined, exec: { agent?: Agent }): string | undefined {
|
|
const sessionCwd = exec.agent?.session.header.cwd
|
|
if (modelWorkdir === undefined) return sessionCwd
|
|
if (sessionCwd !== undefined && !isAbsolute(modelWorkdir)) {
|
|
return resolvePath(sessionCwd, modelWorkdir)
|
|
}
|
|
return modelWorkdir
|
|
}
|
|
|
|
/** Status line for background task reads. */
|
|
function statusLine(task: BashTask): string {
|
|
switch (task.status) {
|
|
case 'running': return '[status: running]'
|
|
case 'killed': return `[status: killed${task.signal !== null ? ` by ${task.signal}` : ''}]`
|
|
case 'completed': return `[status: completed, exit code: ${task.exitCode ?? 0}]`
|
|
}
|
|
}
|
|
|
|
export function apply(ctx: Context): void {
|
|
// The bash tools' cross-call HABIT, which the per-tool descriptions cannot
|
|
// carry (they describe one call each): the exit-code marker is only useful
|
|
// if the model actually checks it every time.
|
|
ctx.systemPrompt.section({
|
|
name: 'tool:bash',
|
|
order: 105,
|
|
text: 'Check the [exit code: N] marker on every bash result; investigate failures before moving on.',
|
|
})
|
|
|
|
/**
|
|
* The caller's owner TOKEN — the owning agent's `session.header.id`, or
|
|
* `undefined` for a non-agent caller. Read `session.header.id` (NOT
|
|
* `session.id`): every other subsystem keys off the header id (the ACP bridge,
|
|
* both persistence backends), and the sibling `resolveWorkdir` already reads
|
|
* `session.header.cwd`, so using `session.id` here would be the asymmetry smell
|
|
* the conventions flag. The two are equal in production, but the header is the
|
|
* canonical identity.
|
|
*/
|
|
const callerToken = (exec: { agent?: Agent }): OwnerToken | undefined =>
|
|
exec.agent ? OwnerToken(exec.agent.session.header.id) : undefined
|
|
|
|
/**
|
|
* Authorize a `bash_output`/`bash_kill` call against the task's stored owner
|
|
* token. Rejects when the task HAS an owner and it differs from the caller's
|
|
* token — using `!== undefined` semantics, NOT truthiness, so an empty-string
|
|
* token is still a real owner (never treated as unowned). An unowned task
|
|
* (`ownerOf` returns `undefined`) is allowed; a truly unknown id is also
|
|
* `undefined` here and then fails loudly at the subsequent
|
|
* `readOutput`/`kill` ("unknown bash task"). The conservative no-agent caller
|
|
* (`callerToken` undefined) cannot match an owned task and is rejected.
|
|
*/
|
|
const assertTaskAccess = (taskId: BashTaskId, exec: { agent?: Agent }): void => {
|
|
const owner = ctx.bash.ownerOf(taskId)
|
|
if (owner !== undefined && owner !== callerToken(exec)) {
|
|
throw new Error(`task ${taskId} belongs to another session`)
|
|
}
|
|
}
|
|
|
|
// Background completion → inject a notice into the owning agent's session.
|
|
// Find the live agent by its session id token via the agent registry, read
|
|
// opportunistically with `ctx.get('agents')` (NOT `ctx.agents`/static inject):
|
|
// this listener runs from `task.done.then` on the bash fiber — a foreign
|
|
// fiber — where the `ctx.agents` property proxy would throw through the
|
|
// traceable shadow; `ctx.get(name)` is the topology-independent lookup. No
|
|
// registry mounted (`undefined`) → drop the notice. Match on
|
|
// `agent.session.header.id`, NOT the registry key: a config agent's id differs
|
|
// from its session id, and the owner token IS the session id.
|
|
ctx.bash.onTaskDone((task) => {
|
|
const ownerToken = ctx.bash.ownerOf(task.id)
|
|
if (ownerToken === undefined) return
|
|
const agent = ctx.get('agents')?.list().find(a => OwnerToken(a.session.header.id) === ownerToken)
|
|
if (!agent) return
|
|
try {
|
|
agent.inject(
|
|
[{ type: 'text', text: `background bash task ${task.id} finished ${statusLine(task)}. Read its output with bash_output.` }],
|
|
{ source: { kind: 'plugin', plugin: 'tool-bash' } },
|
|
)
|
|
} catch (error: unknown) {
|
|
// The ONE expected failure: the agent was disposed between task
|
|
// completion and this injection (ReactLoopAgent.inject throws
|
|
// `agent "<id>" is disposed`). That race is benign — drop the notice.
|
|
// Anything else is a real bug and must surface, not be swallowed.
|
|
if (error instanceof Error && error.message.includes('is disposed')) return
|
|
throw error
|
|
}
|
|
})
|
|
|
|
// The escalation surface exists whenever the mounted executor confines.
|
|
// Its enum is the closed target vocabulary, deliberately NOT cut down by
|
|
// the configured default: a session may switch to a narrower effective mode
|
|
// while sharing this globally registered schema. Strict widening therefore
|
|
// belongs to the per-call check below. An executor swap restarts this fiber
|
|
// (static inject) and re-registers the schema.
|
|
const defaultMode = ctx.bash.sandboxMode
|
|
const escalationModes: readonly SandboxMode[] = defaultMode === undefined ? [] : ESCALATION_TARGETS
|
|
|
|
/**
|
|
* The session's standing mode override for an ordinary (non-escalating)
|
|
* call: the `sandbox/mode` fold of the calling agent's log, stamped
|
|
* onto the request so EXECUTION follows the same effective mode the prompt
|
|
* section states. Weakest precedence — an escalation grant (freshly
|
|
* approved for exactly this call) outranks it, and without either the
|
|
* executor's `resolve()` applies its configured default. Undefined for a
|
|
* non-sandboxing executor (nothing honors it) and for agent-less callers
|
|
* (no session to fold).
|
|
*/
|
|
const sessionOverride = (exec: ToolExecution): SandboxMode | undefined =>
|
|
defaultMode === undefined || exec.agent === undefined ? undefined : effectiveSandboxMode(exec.agent.session.events)
|
|
|
|
/**
|
|
* Resolve a sandbox-escalation request through `ctx.approval` BEFORE
|
|
* anything executes, delegating the shared fail-closed sequence (strict
|
|
* widening, channel resolution, outcome mapping) to
|
|
* {@link approveEscalation}. This tool contributes only the composition
|
|
* guard (the fields are unadvertised without a sandboxing executor, yet
|
|
* schema validation checks advertised keys only, so an unadvertised
|
|
* `sandbox_permissions` still reaches execute) and the channel closure over
|
|
* `ctx.approval` — consumed opportunistically (`ctx.get`, the dsh-tools
|
|
* ask-routing pattern) so a deployment without it degrades per call.
|
|
*/
|
|
const approveBashEscalation = (mode: string, justification: string, exec: ToolExecution): Promise<SandboxMode> => {
|
|
if (escalationModes.length === 0) {
|
|
throw new Error('sandbox_permissions is not available in this composition (no sandboxing executor to escalate)')
|
|
}
|
|
const effectiveMode = (sessionOverride(exec) ?? defaultMode) as SandboxMode
|
|
return approveEscalation(
|
|
{ requestedMode: mode, justification, effectiveMode, subject: 'command' },
|
|
{
|
|
approver: ctx.get('approval'),
|
|
agent: exec.agent,
|
|
callId: exec.callId,
|
|
toolName: 'bash',
|
|
...exec.signal ? { signal: exec.signal } : {},
|
|
},
|
|
)
|
|
}
|
|
|
|
ctx.tools.register(defineTool({
|
|
name: 'bash',
|
|
description: bashDescription(escalationModes),
|
|
parameters: {
|
|
command: { type: 'string', required: true, description: 'The bash command to execute.' },
|
|
description: {
|
|
type: 'string',
|
|
required: true,
|
|
description: 'Clear, concise description of what this command does in active voice, '
|
|
+ '5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; '
|
|
+ '"git status" → "Show working tree status"; "npm install" → "Install package dependencies".',
|
|
},
|
|
timeoutMs: { type: 'number', description: 'Timeout in milliseconds. The executor applies its configured default and cap, and kills the command on expiry.' },
|
|
workdir: { type: 'string', description: 'Working directory for this command. Defaults to the session workspace; a relative path is resolved against it.' },
|
|
run_in_background: { type: 'boolean', description: 'Run in the background and return a task id immediately. No timeout applies.' },
|
|
...escalationModes.length > 0 ? {
|
|
sandbox_permissions: {
|
|
type: 'string' as const,
|
|
enum: [...escalationModes],
|
|
description: 'The wider sandbox mode this command needs. Only valid as a one-shot retry '
|
|
+ 'of a command the sandbox just denied; requires justification and user approval.',
|
|
},
|
|
justification: {
|
|
type: 'string' as const,
|
|
description: 'Required with sandbox_permissions: one sentence for the user explaining '
|
|
+ 'why this exact command needs the wider access.',
|
|
},
|
|
} : {},
|
|
},
|
|
async execute(args: BashToolArgs, exec) {
|
|
validateBashArgs(args)
|
|
// `description` is display/logging metadata only (surfaced to UIs via
|
|
// the tool/call session event); it is intentionally NOT forwarded to
|
|
// ctx.bash and has no effect on execution.
|
|
// An escalating call resolves approval BEFORE anything executes; every
|
|
// non-grant outcome throws its distinct error text and runs nothing.
|
|
// (validateBashArgs pinned the pairing, so the double narrow is exact.)
|
|
// An ordinary call carries the session's standing override instead —
|
|
// grant > session override > executor default (see sessionOverride).
|
|
const sandboxMode = args.sandbox_permissions !== undefined && args.justification !== undefined
|
|
? await approveBashEscalation(args.sandbox_permissions, args.justification, exec)
|
|
: sessionOverride(exec)
|
|
// Default the workdir to the calling agent's session cwd so each ACP
|
|
// session runs in its own workspace (see resolveWorkdir); an explicit
|
|
// model workdir still wins.
|
|
const workdir = resolveWorkdir(args.workdir, exec)
|
|
const request = {
|
|
command: args.command,
|
|
...workdir !== undefined ? { workdir } : {},
|
|
...args.timeoutMs !== undefined ? { timeoutMs: args.timeoutMs } : {},
|
|
...exec.signal ? { signal: exec.signal } : {},
|
|
...sandboxMode !== undefined ? { sandboxMode } : {},
|
|
}
|
|
if (args.run_in_background === true) {
|
|
// Stamp the owner token (the agent's session id) onto the spec so the
|
|
// executor stores it on the task — the isolation fence for bash_output/
|
|
// bash_kill. Foreground runs pass no owner (they finish inline; nothing
|
|
// to fence).
|
|
const task = ctx.bash.start(ctx.bash.resolve({ ...request, owner: callerToken(exec) }))
|
|
return [{ type: 'text', text: `started background task ${task.id}` }]
|
|
}
|
|
const result = await ctx.bash.run(ctx.bash.resolve(request))
|
|
if (result.aborted) throw new Error('command aborted')
|
|
return [{ type: 'text', text: renderResult(result, escalationModes) }]
|
|
},
|
|
presentCall: presentBashCall,
|
|
presentResult: presentBashResult,
|
|
}))
|
|
|
|
ctx.tools.register(defineTool({
|
|
name: 'bash_output',
|
|
description: 'Read new output from a background bash task started with `bash` + `run_in_background`. '
|
|
+ 'Returns only output produced since the previous bash_output call, plus the task status. '
|
|
+ 'Tasks keep running while you do other work; poll again later for more output.',
|
|
parameters: {
|
|
task_id: { type: 'string', required: true, description: 'Task id returned by the bash tool.' },
|
|
},
|
|
// execute is synchronous (registry reads + string shaping) but the
|
|
// ToolDefinition contract wants a Promise — hence resolve(), not async.
|
|
execute(args, exec) {
|
|
const id = validateTaskId(args.task_id)
|
|
assertTaskAccess(id, exec)
|
|
const read = ctx.bash.readOutput(id)
|
|
let text = read.delta.length > 0 ? read.delta : '(no new output)'
|
|
if (read.lossy) {
|
|
const paths = [read.stdoutSpillPath, read.stderrSpillPath].filter((p): p is string => p !== undefined)
|
|
const fullOutput = paths.length > 0 ? paths.join(', ') : '(unavailable)'
|
|
text += `\n[some output was dropped from memory; full output: ${fullOutput}]`
|
|
}
|
|
text += `\n${statusLine(read.task)}`
|
|
if (read.task.sandbox?.runnerFailed) {
|
|
// The sandbox RUNNER itself failed — the command never ran. The
|
|
// foreground path surfaces this as the structured SANDBOX_UNAVAILABLE
|
|
// error; a settled task's read carries the marker instead.
|
|
text += `\n[sandbox: the sandbox runner itself failed under ${read.task.sandbox.mode} mode — the command did not run; this is a sandbox problem, not a command failure]`
|
|
} else if (read.task.sandbox?.denied) {
|
|
// Mirrors the foreground result marker (and its same-turn escalation
|
|
// hint). Background denials are only classifiable once the task
|
|
// settles (the classifier needs the whole stderr), so the marker
|
|
// rides every read that sees the settled task.
|
|
text += `\n${sandboxDenialMarker(read.task.sandbox.mode)}`
|
|
if (escalationModes.length > 0) {
|
|
text += `\n${escalationHintMarker('command')}`
|
|
}
|
|
}
|
|
return Promise.resolve([{ type: 'text', text }])
|
|
},
|
|
presentCall: args => presentTaskCall('Read output from', args),
|
|
}))
|
|
|
|
ctx.tools.register(defineTool({
|
|
name: 'bash_kill',
|
|
description: 'Ask the executor to kill a running background bash task by task id.',
|
|
parameters: {
|
|
task_id: { type: 'string', required: true, description: 'Task id returned by the bash tool.' },
|
|
},
|
|
execute(args, exec) {
|
|
const id = validateTaskId(args.task_id)
|
|
assertTaskAccess(id, exec)
|
|
const killed = ctx.bash.kill(id)
|
|
return Promise.resolve([{
|
|
type: 'text',
|
|
text: killed ? `killed background task ${id}` : `task ${id} had already finished`,
|
|
}])
|
|
},
|
|
presentCall: args => presentTaskCall('Kill', args),
|
|
}))
|
|
}
|