index.ts 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281
  1. /**
  2. * Model-facing `bash` tool over the `ctx.bash` executor seam. Background calls
  3. * register process handles with `ctx.tasks`; their work uses task cancellation
  4. * rather than the tool-call signal after an id is returned.
  5. *
  6. * TODO(permissions): deployment policy belongs in `tools/pre-execute` and
  7. * sandboxing executors; see docs/architecture.md § Extending The Harness.
  8. * @module @deepseek-ai/dsh-tool-bash
  9. */
  10. import type { Context } from 'cordis'
  11. import z from 'schemastery'
  12. import { isAbsolute, resolve as resolvePath } from 'node:path'
  13. import { defineTool } from '@deepseek-ai/dsh-tools'
  14. import type { GenericCallView, TerminalCallView, ToolExecution, ToolResult, ToolResultView } from '@deepseek-ai/dsh-tools'
  15. import type { Agent } from '@deepseek-ai/dsh-agent'
  16. import { assertNever } from '@deepseek-ai/dsh-llm'
  17. import type {} from '@deepseek-ai/dsh-system-prompt'
  18. import type {} from '@deepseek-ai/dsh-tasks'
  19. import type {} from '@deepseek-ai/dsh-user-approval'
  20. import type { SandboxMode } from '@deepseek-ai/dsh-sandbox'
  21. import { effectiveSandboxMode } from '@deepseek-ai/dsh-bash'
  22. import { processOutcome } from './background.ts'
  23. import { parseExitStatus, renderProcessRead, renderResult } from './render.ts'
  24. export const name = 'tool-bash'
  25. export const inject = ['tools', 'bash', 'systemPrompt']
  26. /** Configures whether the model may background commands. */
  27. export interface Config {
  28. /** Expose `run_in_background` (default true); disabled calls are also rejected. */
  29. enableRunInBackground?: boolean
  30. }
  31. export const Config: z<Config> = z.object({
  32. enableRunInBackground: z.boolean().default(true),
  33. })
  34. /** Parsed tool args; execute validates value constraints absent from SchemaSpec. */
  35. interface BashToolArgs {
  36. command: string
  37. description: string
  38. timeoutMs?: number
  39. workdir?: string
  40. run_in_background?: boolean
  41. sandbox_permissions?: string
  42. justification?: string
  43. }
  44. function validateBashArgs(args: BashToolArgs): void {
  45. if (args.command.trim().length === 0) {
  46. throw new Error('invalid command: expected a non-empty string')
  47. }
  48. if (args.description.trim().length === 0) {
  49. throw new Error('invalid description: expected a non-empty string')
  50. }
  51. if (args.timeoutMs !== undefined && (!Number.isFinite(args.timeoutMs) || args.timeoutMs <= 0)) {
  52. throw new Error(`invalid timeoutMs: expected a positive number, got ${JSON.stringify(args.timeoutMs)}`)
  53. }
  54. if (args.sandbox_permissions !== undefined && args.justification === undefined) {
  55. throw new Error('invalid escalation: sandbox_permissions requires a justification')
  56. }
  57. if (args.justification !== undefined && args.sandbox_permissions === undefined) {
  58. throw new Error('invalid escalation: justification is only valid together with sandbox_permissions')
  59. }
  60. if (args.justification !== undefined && args.justification.trim().length === 0) {
  61. throw new Error('invalid justification: expected a non-empty sentence')
  62. }
  63. }
  64. const WIDER_MODES: Record<string, readonly SandboxMode[]> = {
  65. 'read-only': ['workspace-write', 'danger-full-access'],
  66. 'workspace-write': ['danger-full-access'],
  67. }
  68. const ESCALATION_TARGETS: readonly SandboxMode[] = ['workspace-write', 'danger-full-access']
  69. function bashDescription(backgroundEnabled: boolean, escalationModes: readonly SandboxMode[]): string {
  70. const background = backgroundEnabled
  71. ? 'Set `run_in_background: true` for long-running commands: the call returns a task id immediately; read its output with `task_output` and stop it with `task_kill`.'
  72. : 'Background execution is not available; long-running commands must finish within the timeout.'
  73. const base = 'Execute a bash command (`bash -c`) and return its stdout/stderr. '
  74. + 'Each call runs in a fresh shell: no state (cwd, variables, functions) persists between calls — '
  75. + 'pass `workdir` instead of using `cd`. Non-zero exits are reported as `[exit code: N]`. '
  76. + 'Commands may run under a file sandbox; a blocked file operation is reported as `[sandbox: file access denied under <mode> mode]` — a policy denial, not a bug in the command; do not retry another way. '
  77. + 'Long output is truncated to its tail; the full output is saved to a file whose path is reported when available. '
  78. + background
  79. if (escalationModes.length === 0) return base
  80. return base + ' Attempting a command the sandbox may deny is safe and expected: run it and read the '
  81. + 'marker rather than assuming the denial. When a command is denied and a wider mode would let it '
  82. + 'succeed, escalate immediately in the same turn — the one sanctioned exception to a denial: retry '
  83. + 'the exact same command once with `sandbox_permissions` (the narrowest wider mode that suffices) '
  84. + 'plus a one-sentence `justification`. Do not detour through chat to ask permission first — the '
  85. + 'approval prompt raised by that retry is how the user consents. If the session states approval '
  86. + 'prompts are disabled, there is no exception: a denial is final — do not set `sandbox_permissions`. '
  87. + 'Never escalate speculatively: ground the request in a real denial — normally the one this command '
  88. + 'just hit; escalating up front is fine only when this session already denied the same access. '
  89. + 'A rejected escalation is final for that command — stop and explain, never work around '
  90. + 'it — but it does not forbid attempting or escalating other commands later.'
  91. }
  92. /**
  93. * Present foreground calls as terminals and background starts as generic cards.
  94. * The command remains the title on both paths; foreground cwd is passed through
  95. * for the bridge to resolve, while background descriptions remain card content.
  96. */
  97. type BashCallArgs = { command: string; description: string; workdir?: string; run_in_background?: boolean }
  98. function presentBashCall(args: BashCallArgs): GenericCallView | TerminalCallView {
  99. if (args.run_in_background === true) {
  100. return {
  101. card: 'generic',
  102. title: args.command,
  103. kind: 'execute',
  104. rawInput: args.command,
  105. content: [{ type: 'text', text: args.description }],
  106. }
  107. }
  108. return {
  109. card: 'terminal',
  110. title: args.command,
  111. description: args.description,
  112. ...args.workdir !== undefined ? { cwd: args.workdir } : {},
  113. }
  114. }
  115. /**
  116. * Present completed foreground output as a terminal; background acknowledgements
  117. * and execution errors use generic fenced output without an exit-status pill.
  118. */
  119. function presentBashResult(args: unknown, result: ToolResult): ToolResultView | undefined {
  120. const block = result.content.length === 1 ? result.content[0] : undefined
  121. if (block === undefined || block.type !== 'text') return undefined
  122. const raw = block.text
  123. const isBackground = typeof args === 'object' && args !== null && (args as { run_in_background?: unknown }).run_in_background === true
  124. // Background acknowledgements and errors have no terminal exit status.
  125. if (isBackground || result.isError) {
  126. return { card: 'generic', content: [{ type: 'text', text: `\`\`\`console\n${raw.replace(/\n+$/, '')}\n\`\`\`` }] }
  127. }
  128. return { card: 'terminal', output: raw, ...parseExitStatus(raw) }
  129. }
  130. /**
  131. * Resolve an explicit workdir first, making a relative one session-cwd-relative;
  132. * otherwise use the session cwd and leave executor defaulting as the fallback.
  133. */
  134. function resolveWorkdir(modelWorkdir: string | undefined, exec: { agent?: Agent }): string | undefined {
  135. const sessionCwd = exec.agent?.session.header.cwd
  136. if (modelWorkdir === undefined) return sessionCwd
  137. if (sessionCwd !== undefined && !isAbsolute(modelWorkdir)) {
  138. return resolvePath(sessionCwd, modelWorkdir)
  139. }
  140. return modelWorkdir
  141. }
  142. export function apply(ctx: Context, config: Config): void {
  143. const backgroundEnabled = config.enableRunInBackground ?? true
  144. const defaultMode = ctx.bash.sandboxMode
  145. const escalationModes: readonly SandboxMode[] = defaultMode === undefined ? [] : ESCALATION_TARGETS
  146. const sessionOverride = (exec: ToolExecution): SandboxMode | undefined =>
  147. defaultMode === undefined || exec.agent === undefined ? undefined : effectiveSandboxMode(exec.agent.session.events)
  148. const approveEscalation = async (mode: string, justification: string, exec: ToolExecution): Promise<SandboxMode> => {
  149. if (escalationModes.length === 0) {
  150. throw new Error('sandbox_permissions is not available in this composition (no sandboxing executor to escalate)')
  151. }
  152. const effectiveMode = (sessionOverride(exec) ?? defaultMode) as SandboxMode
  153. if (!(WIDER_MODES[effectiveMode] ?? []).includes(mode as SandboxMode)) {
  154. throw new Error(`sandbox escalation to "${mode}" is not strictly wider than this call's current "${effectiveMode}" mode`)
  155. }
  156. const approval = ctx.get('approval')
  157. if (approval === undefined) {
  158. throw new Error(`sandbox escalation to "${mode}" requires approval, but no approval service is composed`)
  159. }
  160. if (exec.agent === undefined) {
  161. throw new Error(`sandbox escalation to "${mode}" requires approval, but the call has no agent to route it through`)
  162. }
  163. const outcome = await approval.request({
  164. agent: exec.agent,
  165. toolName: 'bash',
  166. callId: exec.callId,
  167. reason: `escalate sandbox to ${mode}: ${justification}`,
  168. ...exec.signal ? { signal: exec.signal } : {},
  169. })
  170. switch (outcome) {
  171. case 'allowed-once': return mode as SandboxMode
  172. case 'rejected': throw new Error(`the user rejected escalating this command to "${mode}"`)
  173. case 'cancelled': throw new Error(`approval for escalating to "${mode}" was cancelled`)
  174. case 'unavailable': throw new Error(`sandbox escalation to "${mode}" requires approval, but no approval channel is available`)
  175. default: return assertNever(outcome, 'ApprovalOutcome')
  176. }
  177. }
  178. // Cross-call guidance belongs in the prompt rather than one-call schema prose.
  179. ctx.systemPrompt.section({
  180. name: 'tool:bash',
  181. order: 105,
  182. text: 'Check the [exit code: N] marker on every bash result; investigate failures before moving on.',
  183. })
  184. ctx.tools.register(defineTool({
  185. name: 'bash',
  186. description: bashDescription(backgroundEnabled, escalationModes),
  187. parameters: {
  188. command: { type: 'string', required: true, description: 'The bash command to execute.' },
  189. description: {
  190. type: 'string',
  191. required: true,
  192. description: 'Clear, concise description of what this command does in active voice, '
  193. + '5-10 words (shown in the UI). Examples: "ls" → "List files in current directory"; '
  194. + '"git status" → "Show working tree status"; "npm install" → "Install package dependencies".',
  195. },
  196. timeoutMs: { type: 'number', description: 'Timeout in milliseconds. The executor applies its configured default and cap, and kills the command on expiry.' },
  197. workdir: { type: 'string', description: 'Working directory for this command. Defaults to the session workspace; a relative path is resolved against it.' },
  198. ...backgroundEnabled ? {
  199. run_in_background: { type: 'boolean' as const, description: 'Run in the background and return a task id immediately (collect with task_output, stop with task_kill). No timeout applies.' },
  200. } : {},
  201. ...escalationModes.length > 0 ? {
  202. sandbox_permissions: {
  203. type: 'string' as const,
  204. enum: [...escalationModes],
  205. description: 'The wider sandbox mode this command needs. Only valid as a one-shot retry of a command the sandbox just denied; requires justification and user approval.',
  206. },
  207. justification: {
  208. type: 'string' as const,
  209. description: 'Required with sandbox_permissions: one sentence for the user explaining why this exact command needs the wider access.',
  210. },
  211. } : {},
  212. },
  213. async execute(args: BashToolArgs, exec) {
  214. validateBashArgs(args)
  215. // Description is display metadata; workdir defaults to the caller's session.
  216. const sandboxMode = args.sandbox_permissions !== undefined && args.justification !== undefined
  217. ? await approveEscalation(args.sandbox_permissions, args.justification, exec)
  218. : sessionOverride(exec)
  219. const workdir = resolveWorkdir(args.workdir, exec)
  220. const request = {
  221. command: args.command,
  222. ...workdir !== undefined ? { workdir } : {},
  223. ...args.timeoutMs !== undefined ? { timeoutMs: args.timeoutMs } : {},
  224. ...sandboxMode !== undefined ? { sandboxMode } : {},
  225. }
  226. if (args.run_in_background === true) {
  227. // Undeclared keys are allowed, so schema omission also needs enforcement.
  228. if (!backgroundEnabled) {
  229. throw new Error('run_in_background is disabled for this deployment (enableRunInBackground: false)')
  230. }
  231. const tasks = ctx.get('tasks')
  232. if (tasks === undefined) {
  233. throw new Error('background tasks unavailable: load @deepseek-ai/dsh-tasks and @deepseek-ai/dsh-tool-tasks')
  234. }
  235. // Reject pre-start cancellation; returned tasks use their own lifecycle.
  236. if (exec.signal?.aborted) throw new Error('command aborted')
  237. // Task preflight finishes before the starter can spawn a process.
  238. const id = tasks.start({
  239. kind: 'bash',
  240. label: args.command,
  241. ...exec.agent ? { owner: exec.agent } : {},
  242. run: () => {
  243. const proc = ctx.bash.start(ctx.bash.resolve(request))
  244. return {
  245. cancel: () => void proc.kill(),
  246. done: proc.done.then(() => processOutcome(proc)),
  247. readOutput: () => renderProcessRead(proc.readOutput(), proc.sandbox, escalationModes),
  248. }
  249. },
  250. })
  251. return [{ type: 'text', text: `started background task ${id}` }]
  252. }
  253. const result = await ctx.bash.run(ctx.bash.resolve({
  254. ...request,
  255. ...exec.signal ? { signal: exec.signal } : {},
  256. }))
  257. if (result.aborted) throw new Error('command aborted')
  258. return [{ type: 'text', text: renderResult(result, escalationModes) }]
  259. },
  260. presentCall: presentBashCall,
  261. presentResult: presentBashResult,
  262. }))
  263. }