395 lines
18 KiB
TypeScript
395 lines
18 KiB
TypeScript
/**
|
|
* Model-facing Consumer of the `ctx.bash` capability seam. Background calls
|
|
* register process handles with `ctx.tasks`; their work uses task cancellation
|
|
* rather than the tool-call signal after an id is returned.
|
|
*
|
|
* TODO(permissions): deployment policy belongs in `tools/pre-execute` and
|
|
* sandboxing executors; see docs/architecture.md § Extending The Harness.
|
|
* @module @deepseek-ai/dsh-tool-bash
|
|
*/
|
|
|
|
import type { Context } from 'cordis'
|
|
import z from 'schemastery'
|
|
import { isAbsolute, resolve as resolvePath } from 'node:path'
|
|
import { defineTool, TOOL_ABORTED } from '@deepseek-ai/dsh-tools'
|
|
import type { GenericCallView, TerminalCallView, ToolExecution, ToolResult, ToolResultView } from '@deepseek-ai/dsh-tools'
|
|
import { HarnessError } from '@deepseek-ai/dsh-llm'
|
|
import type { Agent } from '@deepseek-ai/dsh-agent'
|
|
import type {} from '@deepseek-ai/dsh-system-prompt'
|
|
import type {} from '@deepseek-ai/dsh-tasks'
|
|
import type {} from '@deepseek-ai/dsh-user-approval'
|
|
import type {} from '@deepseek-ai/dsh-bash-env'
|
|
import type { SandboxExecutionPolicy, SandboxMode } from '@deepseek-ai/dsh-sandbox'
|
|
import { ESCALATION_TARGETS, approveEscalation, canonicalPath, validateEscalationArgs } from '@deepseek-ai/dsh-sandbox'
|
|
import type { SandboxPolicyService } from '@deepseek-ai/dsh-sandbox-policy'
|
|
import { DSH_ENV_PREFIX } from '@deepseek-ai/dsh-bash'
|
|
import type { BashRunResult } from '@deepseek-ai/dsh-bash'
|
|
import { processOutcome } from './background.ts'
|
|
import { parseExitStatus, renderProcessRead, renderResult } from './render.ts'
|
|
|
|
export const name = 'tool-bash'
|
|
export const inject = ['tools', 'bash', 'systemPrompt', 'bashEnv']
|
|
|
|
/** Configuration for the bash tool. */
|
|
export interface Config {
|
|
/** Expose `run_in_background` (default true); disabled calls are also rejected. */
|
|
enableRunInBackground?: boolean
|
|
}
|
|
|
|
/** Runtime configuration schema for the bash tool plugin. */
|
|
export const Config: z<Config> = z.object({
|
|
enableRunInBackground: z.boolean().default(true),
|
|
})
|
|
|
|
/** Parsed tool args; execute validates value constraints absent from ParameterSchemaSpec. */
|
|
interface BashToolArgs {
|
|
command: string
|
|
description: string
|
|
timeoutMs?: number
|
|
workdir?: string
|
|
run_in_background?: boolean
|
|
sandbox_permissions?: string
|
|
justification?: string
|
|
}
|
|
|
|
function validateBashArgs(args: BashToolArgs): void {
|
|
if (args.command.trim().length === 0) {
|
|
throw new Error('invalid command: expected a non-empty string')
|
|
}
|
|
if (args.description.trim().length === 0) {
|
|
throw new Error('invalid description: expected a non-empty string')
|
|
}
|
|
if (args.timeoutMs !== undefined && (!Number.isFinite(args.timeoutMs) || args.timeoutMs <= 0)) {
|
|
throw new Error(`invalid timeoutMs: expected a positive number, got ${JSON.stringify(args.timeoutMs)}`)
|
|
}
|
|
// The escalation pairing (sandbox_permissions ⇔ justification, non-empty) is
|
|
// the shared rule both enforcing families validate identically.
|
|
validateEscalationArgs(args.sandbox_permissions, args.justification)
|
|
}
|
|
|
|
function bashDescription(backgroundEnabled: boolean, escalationModes: readonly SandboxMode[]): string {
|
|
const background = backgroundEnabled
|
|
? 'Set `run_in_background: true` for long-running commands: the call returns a task id immediately; read its output with `task_output` and stop it with `task_kill`.'
|
|
: 'Background execution is not available; long-running commands must finish within the timeout.'
|
|
const base = 'Execute a bash command (`bash -c`) and return its stdout/stderr. '
|
|
+ 'Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — '
|
|
+ 'pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. '
|
|
+ `Current harness environment facts are exposed through managed \`$${DSH_ENV_PREFIX}*\` variables; inspect them when needed. `
|
|
+ 'Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way. '
|
|
+ 'Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. '
|
|
+ background
|
|
if (escalationModes.length === 0) return base
|
|
return base + ' Attempting a command the sandbox may deny is safe and expected: run it and read the '
|
|
+ 'marker rather than assuming the denial. When a command is denied and a wider mode would let it '
|
|
+ 'succeed, escalate immediately in the same turn — the one sanctioned exception to a denial: retry '
|
|
+ 'the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) '
|
|
+ 'plus a one-sentence `justification`. Do not detour through chat to ask permission first — the '
|
|
+ 'approval prompt raised by that retry is how the user consents. If the session states approval '
|
|
+ 'prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. '
|
|
+ 'Never escalate speculatively: ground the request in a real denial — normally the one this command '
|
|
+ 'just hit; escalating up front is fine only when this session already denied the same access. '
|
|
+ 'A rejected escalation is final for that command — stop and explain, never work around '
|
|
+ 'it — but it does not forbid attempting or escalating other commands later.'
|
|
}
|
|
|
|
/**
|
|
* Present foreground calls as terminals and background starts as generic cards.
|
|
* The command remains the title on both paths; foreground cwd is passed through
|
|
* for the bridge to resolve, while background descriptions remain card content.
|
|
*/
|
|
type BashCallArgs = { command: string; description: string; workdir?: string; run_in_background?: boolean }
|
|
|
|
function presentBashCall(args: BashCallArgs): GenericCallView | TerminalCallView {
|
|
if (args.run_in_background === true) {
|
|
return {
|
|
card: 'generic',
|
|
title: args.command,
|
|
kind: 'execute',
|
|
rawInput: args.command,
|
|
content: [{ type: 'text', text: args.description }],
|
|
}
|
|
}
|
|
return {
|
|
card: 'terminal',
|
|
title: args.command,
|
|
description: args.description,
|
|
...args.workdir !== undefined ? { cwd: args.workdir } : {},
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Present completed foreground output as a terminal; background acknowledgements
|
|
* and execution errors use generic fenced output without an exit-status pill.
|
|
*/
|
|
function presentBashResult(args: unknown, result: ToolResult): ToolResultView | undefined {
|
|
const block = result.content.length === 1 ? result.content[0] : undefined
|
|
if (block === undefined || block.type !== 'text') return undefined
|
|
const raw = block.text
|
|
const isBackground = typeof args === 'object' && args !== null && (args as { run_in_background?: unknown }).run_in_background === true
|
|
// Background acknowledgements and errors have no terminal exit status.
|
|
if (isBackground || result.isError) {
|
|
return { card: 'generic', content: [{ type: 'text', text: `\`\`\`console\n${raw.replace(/\n+$/, '')}\n\`\`\`` }] }
|
|
}
|
|
// The exit marker becomes the card's exit pill, so it leaves the output body.
|
|
const { body, ...exit } = parseExitStatus(raw)
|
|
return { card: 'terminal', output: body, ...exit }
|
|
}
|
|
|
|
/**
|
|
* Resolve an explicit workdir first, making a relative one session-workspace-relative;
|
|
* otherwise use the filesystem identity of the session cwd and leave executor
|
|
* defaulting as the fallback. A resolved sandbox-policy root wins so workdir
|
|
* and confinement use the exact same per-call identity.
|
|
*/
|
|
function resolveWorkdir(
|
|
modelWorkdir: string | undefined,
|
|
exec: { agent?: Agent },
|
|
policyWorkspaceRoot?: string,
|
|
): string | undefined {
|
|
const headerCwd = exec.agent?.session.header.cwd
|
|
const sessionCwd = policyWorkspaceRoot ?? (headerCwd === undefined ? undefined : canonicalPath(headerCwd))
|
|
if (modelWorkdir === undefined) return sessionCwd
|
|
if (sessionCwd !== undefined && !isAbsolute(modelWorkdir)) {
|
|
return resolvePath(sessionCwd, modelWorkdir)
|
|
}
|
|
return modelWorkdir
|
|
}
|
|
|
|
/** Detach the executor DTO from readonly Service Definition types into plain JSON data. */
|
|
function canonicalBashResult(result: BashRunResult) {
|
|
const output = (stream: BashRunResult['stdout']) => ({
|
|
text: stream.text,
|
|
truncated: stream.truncated,
|
|
...stream.spillPath !== undefined ? { spillPath: stream.spillPath } : {},
|
|
})
|
|
return {
|
|
exitCode: result.exitCode,
|
|
signal: result.signal,
|
|
timedOut: result.timedOut,
|
|
aborted: result.aborted,
|
|
timeoutMs: result.timeoutMs,
|
|
stdout: output(result.stdout),
|
|
stderr: output(result.stderr),
|
|
...result.sandbox !== undefined ? {
|
|
sandbox: {
|
|
mode: result.sandbox.mode,
|
|
denied: result.sandbox.denied,
|
|
...result.sandbox.enforcement !== undefined ? { enforcement: result.sandbox.enforcement } : {},
|
|
...result.sandbox.runnerFailed !== undefined ? { runnerFailed: result.sandbox.runnerFailed } : {},
|
|
},
|
|
} : {},
|
|
}
|
|
}
|
|
|
|
/** Canonical background-handle properties shared by the bash output union. */
|
|
const BACKGROUND_OUTPUT_PROPERTIES = {
|
|
kind: { type: 'string', required: true, const: 'background' },
|
|
taskId: { type: 'string', required: true },
|
|
} as const
|
|
|
|
export function apply(ctx: Context, config: Config = {}): void {
|
|
const backgroundEnabled = config.enableRunInBackground ?? true
|
|
const defaultMode = ctx.bash.sandboxMode
|
|
const escalationModes: readonly SandboxMode[] = defaultMode === undefined ? [] : ESCALATION_TARGETS
|
|
const sandboxPolicy: SandboxPolicyService | undefined = defaultMode === undefined ? undefined : ctx.get('sandboxPolicy')
|
|
if (defaultMode !== undefined && sandboxPolicy === undefined) {
|
|
throw new Error('tool-bash: the mounted bash executor confines but ctx.sandboxPolicy is missing')
|
|
}
|
|
/** Resolve the complete standing policy for this call when a confining executor is mounted. */
|
|
const resolveSandboxPolicy = (exec: ToolExecution): SandboxExecutionPolicy | undefined =>
|
|
sandboxPolicy?.resolve(exec.agent === undefined ? {} : { session: exec.agent.session })
|
|
|
|
/**
|
|
* Resolve a sandbox-escalation request through `ctx.approval` BEFORE
|
|
* anything executes, delegating the shared fail-closed sequence (strict
|
|
* widening, channel resolution, outcome mapping) to
|
|
* {@link approveEscalation}. This tool contributes only the composition
|
|
* guard (the fields are unadvertised without a sandboxing executor, yet
|
|
* schema validation checks advertised keys only, so an unadvertised
|
|
* `sandbox_permissions` still reaches execute) and the approval ingredients
|
|
* The shared policy resolver is required whenever the executor advertises
|
|
* confinement, so a split composition fails at tool-plugin load.
|
|
*/
|
|
const approveBashEscalation = (
|
|
mode: string,
|
|
justification: string,
|
|
exec: ToolExecution,
|
|
standingPolicy: SandboxExecutionPolicy | undefined,
|
|
): Promise<SandboxMode> => {
|
|
if (escalationModes.length === 0) {
|
|
throw new Error('sandbox_permissions is not available in this composition (no sandboxing executor to escalate)')
|
|
}
|
|
const effectiveMode = (standingPolicy as SandboxExecutionPolicy).mode
|
|
return approveEscalation(
|
|
{ requestedMode: mode, justification, effectiveMode, subject: 'command' },
|
|
{
|
|
approver: ctx.get('approval'),
|
|
agent: exec.agent,
|
|
callId: exec.callId,
|
|
toolName: 'bash',
|
|
signal: exec.signal,
|
|
},
|
|
)
|
|
}
|
|
|
|
// Cross-call guidance belongs in the prompt rather than one-call schema prose.
|
|
ctx.systemPrompt.section({
|
|
name: 'tool:bash',
|
|
order: 105,
|
|
text: 'Check the [exit code: N] marker on every bash result; investigate failures before moving on.',
|
|
})
|
|
|
|
ctx.tools.register(defineTool({
|
|
name: 'bash',
|
|
description: bashDescription(backgroundEnabled, escalationModes),
|
|
parameters: {
|
|
command: { type: 'string', required: true, description: 'The bash command to execute.' },
|
|
description: {
|
|
type: 'string',
|
|
required: true,
|
|
description: 'Clear, concise description of what this command does in active voice, '
|
|
+ '5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; '
|
|
+ '"git status" → "Show working tree status"; "npm install" → "Install package dependencies".',
|
|
},
|
|
timeoutMs: { type: 'number', description: 'Timeout in milliseconds. The executor applies its configured default and cap, and kills the command on expiry.' },
|
|
workdir: { type: 'string', description: 'Working directory for this command. Defaults to the session workspace; a relative path is resolved against it.' },
|
|
...backgroundEnabled ? {
|
|
run_in_background: { type: 'boolean' as const, description: 'Run in the background and return a task id immediately (collect with task_output, stop with task_kill). No timeout applies.' },
|
|
} : {},
|
|
...escalationModes.length > 0 ? {
|
|
sandbox_permissions: {
|
|
type: 'string' as const,
|
|
enum: [...escalationModes],
|
|
description: 'The wider sandbox mode this command needs. Only valid as a one-shot retry of a command the sandbox just denied; requires justification and user approval.',
|
|
},
|
|
justification: {
|
|
type: 'string' as const,
|
|
description: 'Required with sandbox_permissions: one sentence for the user explaining why this exact command needs the wider access.',
|
|
},
|
|
} : {},
|
|
},
|
|
output: {
|
|
schema: {
|
|
oneOf: [
|
|
{
|
|
type: 'object',
|
|
additionalProperties: false,
|
|
properties: BACKGROUND_OUTPUT_PROPERTIES,
|
|
},
|
|
{
|
|
type: 'object',
|
|
additionalProperties: false,
|
|
properties: {
|
|
kind: { type: 'string', required: true, const: 'foreground' },
|
|
exitCode: { required: true, oneOf: [{ type: 'integer' }, { type: 'null' }] },
|
|
signal: { required: true, oneOf: [{ type: 'string' }, { type: 'null' }] },
|
|
timedOut: { type: 'boolean', required: true },
|
|
aborted: { type: 'boolean', required: true },
|
|
timeoutMs: { type: 'number', required: true },
|
|
stdout: {
|
|
type: 'object',
|
|
additionalProperties: false,
|
|
required: true,
|
|
properties: {
|
|
text: { type: 'string', required: true },
|
|
truncated: { type: 'boolean', required: true },
|
|
spillPath: { type: 'string' },
|
|
},
|
|
},
|
|
stderr: {
|
|
type: 'object',
|
|
additionalProperties: false,
|
|
required: true,
|
|
properties: {
|
|
text: { type: 'string', required: true },
|
|
truncated: { type: 'boolean', required: true },
|
|
spillPath: { type: 'string' },
|
|
},
|
|
},
|
|
sandbox: {
|
|
type: 'object',
|
|
additionalProperties: false,
|
|
properties: {
|
|
mode: { type: 'string', required: true },
|
|
denied: { type: 'boolean', required: true },
|
|
enforcement: { type: 'string' },
|
|
runnerFailed: { type: 'boolean' },
|
|
},
|
|
},
|
|
},
|
|
},
|
|
],
|
|
},
|
|
render: (_args, value) => [{
|
|
type: 'text',
|
|
text: value.kind === 'background'
|
|
? `started background task ${value.taskId}`
|
|
: renderResult(value as { kind: 'foreground' } & BashRunResult, escalationModes),
|
|
}],
|
|
},
|
|
async execute(args: BashToolArgs, exec) {
|
|
validateBashArgs(args)
|
|
// Description is display metadata; workdir defaults to the caller's session.
|
|
const standingPolicy = resolveSandboxPolicy(exec)
|
|
const approvedMode = args.sandbox_permissions !== undefined && args.justification !== undefined
|
|
? await approveBashEscalation(args.sandbox_permissions, args.justification, exec, standingPolicy)
|
|
: undefined
|
|
const policy = approvedMode === undefined
|
|
? standingPolicy
|
|
: { ...(standingPolicy as SandboxExecutionPolicy), mode: approvedMode }
|
|
const workdir = resolveWorkdir(args.workdir, exec, standingPolicy?.workspaceRoot)
|
|
const dshEnv = ctx.bashEnv.collect(exec)
|
|
const request = {
|
|
command: args.command,
|
|
...workdir !== undefined ? { workdir } : {},
|
|
...args.timeoutMs !== undefined ? { timeoutMs: args.timeoutMs } : {},
|
|
dshEnv,
|
|
...policy !== undefined ? { sandboxPolicy: policy } : {},
|
|
}
|
|
if (args.run_in_background === true) {
|
|
// Undeclared keys are allowed, so schema omission also needs enforcement.
|
|
if (!backgroundEnabled) {
|
|
throw new Error('run_in_background is disabled for this deployment (enableRunInBackground: false)')
|
|
}
|
|
const tasks = ctx.get('tasks')
|
|
if (tasks === undefined) {
|
|
throw new Error('background tasks unavailable: load @deepseek-ai/dsh-tasks and @deepseek-ai/dsh-tool-tasks')
|
|
}
|
|
// The caller owns cancellation until ctx.tasks commits detached ownership.
|
|
if (exec.signal.aborted) {
|
|
const error = new HarnessError('tool call aborted', TOOL_ABORTED)
|
|
error.name = 'AbortError'
|
|
throw error
|
|
}
|
|
// Task preflight finishes before the starter can spawn a process.
|
|
const id = tasks.start({
|
|
kind: 'bash',
|
|
label: args.command,
|
|
...exec.agent ? { owner: exec.agent } : {},
|
|
run: () => {
|
|
const proc = ctx.bash.start(ctx.bash.resolve(request))
|
|
return {
|
|
cancel: () => void proc.kill(),
|
|
done: proc.done.then(() => processOutcome(proc)),
|
|
readOutput: () => renderProcessRead(proc.readOutput(), proc.sandbox, escalationModes),
|
|
}
|
|
},
|
|
})
|
|
return { kind: 'background' as const, taskId: id }
|
|
}
|
|
const result = await ctx.bash.run(ctx.bash.resolve({
|
|
...request,
|
|
signal: exec.signal,
|
|
}))
|
|
if (result.aborted) {
|
|
const error = new HarnessError('tool call aborted', TOOL_ABORTED)
|
|
error.name = 'AbortError'
|
|
throw error
|
|
}
|
|
return { kind: 'foreground' as const, ...canonicalBashResult(result) }
|
|
},
|
|
presentCall: presentBashCall,
|
|
presentResult: presentBashResult,
|
|
}))
|
|
}
|