tools.spec.ts 52 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000100110021003100410051006100710081009101010111012101310141015101610171018101910201021102210231024102510261027102810291030103110321033103410351036103710381039104010411042104310441045104610471048104910501051105210531054105510561057105810591060106110621063106410651066106710681069107010711072107310741075107610771078107910801081108210831084108510861087108810891090109110921093109410951096109710981099110011011102110311041105110611071108110911101111111211131114111511161117111811191120112111221123112411251126112711281129113011311132113311341135113611371138113911401141114211431144114511461147114811491150115111521153115411551156115711581159116011611162116311641165116611671168116911701171117211731174117511761177117811791180118111821183118411851186118711881189
  1. import { mkdtempSync } from 'node:fs'
  2. import { tmpdir } from 'node:os'
  3. import { join } from 'node:path'
  4. import { describe, expect, it, vi } from 'vitest'
  5. import { Context } from 'cordis'
  6. import { CallId } from '@deepseek-ai/dsh-llm'
  7. import { BashExecutor } from '@deepseek-ai/dsh-bash'
  8. import type { BashExecRequest, BashExecSpec, BashProcess, BashProcessRead, BashRunResult } from '@deepseek-ai/dsh-bash'
  9. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  10. import ToolRegistry from '@deepseek-ai/dsh-tools'
  11. import AgentRegistry from '@deepseek-ai/dsh-agent'
  12. import type { Agent } from '@deepseek-ai/dsh-agent'
  13. import SessionStore, { SessionId } from '@deepseek-ai/dsh-session'
  14. import SessionPersistenceJsonl from '@deepseek-ai/dsh-session-persistence-jsonl'
  15. import TaskService from '@deepseek-ai/dsh-tasks'
  16. import * as ToolTasks from '@deepseek-ai/dsh-tool-tasks'
  17. import ApprovalService from '@deepseek-ai/dsh-user-approval'
  18. import type { ApprovalOutcome } from '@deepseek-ai/dsh-user-approval'
  19. import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local'
  20. import * as ToolBash from '@deepseek-ai/dsh-tool-bash'
  21. import { processOutcome } from '../src/background.ts'
  22. import { renderProcessRead, renderResult } from '../src/render.ts'
  23. const spillDir = mkdtempSync(join(tmpdir(), 'dsh-tool-bash-spec-'))
  24. /** Foreground-only harness: no task runtime (backgrounding fails loud here). */
  25. async function setup() {
  26. const ctx = new Context()
  27. await ctx.plugin(SystemPrompt)
  28. await ctx.plugin(ToolRegistry)
  29. await ctx.plugin(AgentRegistry)
  30. await ctx.plugin(LocalBashExecutor, { timeoutMs: 10_000, graceMs: 200 })
  31. ;(ctx.bash as LocalBashExecutor).internals = { spillDir }
  32. await ctx.plugin(ToolBash)
  33. return ctx
  34. }
  35. /** Full harness: the generic task runtime + its control surface, then the bash tool. */
  36. async function setupWithTasks() {
  37. const ctx = new Context()
  38. await ctx.plugin(SystemPrompt)
  39. await ctx.plugin(ToolRegistry)
  40. await ctx.plugin(AgentRegistry)
  41. await ctx.plugin(TaskService)
  42. await ctx.plugin(ToolTasks)
  43. await ctx.plugin(LocalBashExecutor, { timeoutMs: 10_000, graceMs: 200 })
  44. ;(ctx.bash as LocalBashExecutor).internals = { spillDir }
  45. await ctx.plugin(ToolBash)
  46. return ctx
  47. }
  48. /**
  49. * Build a fake {@link Agent} with the shared agent/session identity, give it a
  50. * dedicated lifecycle fiber for `Agent.ctx`, and register it in `ctx.agents`.
  51. */
  52. function registerFakeAgent(ctx: Context, sessionId: string, inject: (...args: unknown[]) => void = () => {}): Agent {
  53. const scopeFiber = ctx.plugin(() => {})
  54. const id = SessionId(sessionId)
  55. const agent = {
  56. id,
  57. ctx: scopeFiber.ctx,
  58. inject,
  59. session: { id, header: { version: 0, id, createdAt: 0 } },
  60. } as unknown as Agent
  61. ctx.agents.register(agent)
  62. return agent
  63. }
  64. let callCounter = 0
  65. function call(ctx: Context, name: string, args: unknown, agent?: Agent) {
  66. return ctx.tools.execute({ callId: CallId(`call-${++callCounter}`), name, arguments: args, ...agent ? { agent } : {} })
  67. }
  68. function text(result: { content: { type: string; text?: string }[] }): string {
  69. return result.content.filter(block => block.type === 'text').map(block => block.text).join('')
  70. }
  71. async function callUntilText(
  72. ctx: Context,
  73. name: string,
  74. args: unknown,
  75. expected: string,
  76. timeoutMs = 5_000,
  77. ): Promise<Awaited<ReturnType<typeof call>>> {
  78. const deadline = Date.now() + timeoutMs
  79. let last: Awaited<ReturnType<typeof call>> | undefined
  80. while (Date.now() < deadline) {
  81. last = await call(ctx, name, args)
  82. if (text(last).includes(expected)) return last
  83. await new Promise(resolve => setTimeout(resolve, 20))
  84. }
  85. throw new Error(`${name} output did not include ${JSON.stringify(expected)}; last text was ${JSON.stringify(last !== undefined ? text(last) : '')}`)
  86. }
  87. class RecordingSandboxExecutor extends BashExecutor {
  88. readonly modes: Array<string | undefined> = []
  89. override get sandboxMode() {
  90. return 'read-only' as const
  91. }
  92. resolve(request: BashExecRequest): BashExecSpec {
  93. return {
  94. command: request.command,
  95. workdir: request.workdir ?? process.cwd(),
  96. stdoutMaxBytes: request.stdoutMaxBytes ?? 64_000,
  97. timeoutMs: request.timeoutMs ?? 1000,
  98. ...request.signal ? { signal: request.signal } : {},
  99. sandboxMode: request.sandboxMode ?? 'read-only',
  100. }
  101. }
  102. run(spec: BashExecSpec): Promise<BashRunResult> {
  103. this.modes.push(spec.sandboxMode)
  104. return Promise.resolve({
  105. exitCode: 0,
  106. signal: null,
  107. timedOut: false,
  108. aborted: false,
  109. timeoutMs: spec.timeoutMs,
  110. stdout: { text: 'ok', truncated: false },
  111. stderr: { text: '', truncated: false },
  112. sandbox: {
  113. mode: spec.sandboxMode ?? 'read-only',
  114. denied: false,
  115. ...spec.command === 'without optional sandbox facts'
  116. ? {}
  117. : { enforcement: 'full' as const, runnerFailed: false },
  118. },
  119. })
  120. }
  121. start(spec: BashExecSpec): BashProcess {
  122. this.modes.push(spec.sandboxMode)
  123. return {
  124. status: 'completed',
  125. exitCode: 0,
  126. signal: null,
  127. done: Promise.resolve(),
  128. sandbox: { mode: spec.sandboxMode ?? 'read-only', denied: false },
  129. readOutput: () => ({ delta: '', lossy: false }),
  130. kill: () => false,
  131. }
  132. }
  133. }
  134. /** Test executor that records whether the background start boundary was crossed. */
  135. class CountingStartExecutor extends BashExecutor {
  136. starts = 0
  137. resolve(request: BashExecRequest): BashExecSpec {
  138. return {
  139. command: request.command,
  140. workdir: request.workdir ?? '/x',
  141. timeoutMs: request.timeoutMs ?? 0,
  142. stdoutMaxBytes: request.stdoutMaxBytes ?? 64_000,
  143. sandboxMode: request.sandboxMode,
  144. }
  145. }
  146. run(): Promise<BashRunResult> { return Promise.reject(new Error('unused')) }
  147. start(): BashProcess {
  148. this.starts += 1
  149. return {
  150. status: 'completed',
  151. exitCode: 0,
  152. signal: null,
  153. done: Promise.resolve(),
  154. readOutput: () => ({ delta: '', lossy: false }),
  155. kill: () => false,
  156. }
  157. }
  158. }
  159. async function setupSandboxed(withApproval = false) {
  160. const ctx = new Context()
  161. await ctx.plugin(SystemPrompt)
  162. await ctx.plugin(ToolRegistry)
  163. await ctx.plugin(AgentRegistry)
  164. await ctx.plugin(TaskService)
  165. await ctx.plugin(ToolTasks)
  166. await ctx.plugin(RecordingSandboxExecutor)
  167. if (withApproval) await ctx.plugin(ApprovalService)
  168. await ctx.plugin(ToolBash)
  169. return { ctx, bash: ctx.bash as RecordingSandboxExecutor }
  170. }
  171. function sandboxAgent(mode?: 'read-only' | 'workspace-write' | 'danger-full-access', ctx?: Context): Agent {
  172. const events: Array<{ type: string; data?: Record<string, unknown> }> = [{ type: 'turn/start' }]
  173. if (mode !== undefined) events.push({ type: 'sandbox/mode', data: { mode } })
  174. const id = SessionId('sandbox-session')
  175. return {
  176. id,
  177. ...ctx === undefined ? {} : { ctx: ctx.plugin(() => {}).ctx },
  178. session: {
  179. id,
  180. header: { version: 0, id, createdAt: 0 },
  181. events,
  182. append: (type: string, data: Record<string, unknown>) => {
  183. const event = { type, data }
  184. events.push(event)
  185. return event
  186. },
  187. },
  188. } as unknown as Agent
  189. }
  190. describe('bash tool', () => {
  191. it('returns stdout for a successful command', async () => {
  192. const ctx = await setup()
  193. const result = await call(ctx, 'bash', { command: 'echo hello', description: 'test command' })
  194. expect(result.isError).toBe(false)
  195. if (result.isError) throw new Error('expected bash success')
  196. expect(result.value).toMatchObject({
  197. kind: 'foreground',
  198. exitCode: 0,
  199. signal: null,
  200. timedOut: false,
  201. aborted: false,
  202. stdout: { text: 'hello\n', truncated: false },
  203. stderr: { text: '', truncated: false },
  204. })
  205. expect(text(result)).toBe('hello\n')
  206. })
  207. it('reports (no output) for silent commands', async () => {
  208. const ctx = await setup()
  209. const result = await call(ctx, 'bash', { command: 'true', description: 'test command' })
  210. expect(text(result)).toBe('(no output)')
  211. })
  212. it('marks stderr sections', async () => {
  213. const ctx = await setup()
  214. const result = await call(ctx, 'bash', { command: 'echo out; echo err >&2', description: 'test command' })
  215. expect(text(result)).toBe('out\n[stderr]\nerr\n')
  216. expect(result.isError).toBe(false)
  217. })
  218. it('reports non-zero exits without isError', async () => {
  219. const ctx = await setup()
  220. const result = await call(ctx, 'bash', { command: 'echo failing; exit 3', description: 'test command' })
  221. expect(result.isError).toBe(false)
  222. expect(text(result)).toBe('failing\n[exit code: 3]')
  223. })
  224. it('reports timeout kills with both markers (timeout first)', async () => {
  225. const ctx = await setup()
  226. const result = await call(ctx, 'bash', { command: 'sleep 60', description: 'test command', timeoutMs: 100 })
  227. expect(result.isError).toBe(false)
  228. expect(text(result)).toBe('(no output)\n[timed out after 100ms]\n[killed by signal: SIGTERM]')
  229. })
  230. it('reports a timeout even when the command traps the signal and exits 0', async () => {
  231. // The signal-independent timeout marker: a trapped SIGTERM that exits 0
  232. // after our timer fired must NOT look like a clean success. (bash may
  233. // print "Terminated" to stderr for the killed sleep — environment
  234. // dependent — so assert the marker, not the exact body.)
  235. const ctx = await setup()
  236. const result = await call(ctx, 'bash', { command: 'trap "exit 0" TERM; sleep 60', description: 'test command', timeoutMs: 100 })
  237. expect(result.isError).toBe(false)
  238. expect(text(result)).toContain('[timed out after 100ms]')
  239. expect(text(result)).not.toContain('[exit code:')
  240. })
  241. it('reports truncation with the spill path', async () => {
  242. const ctx = new Context()
  243. await ctx.plugin(SystemPrompt)
  244. await ctx.plugin(ToolRegistry)
  245. await ctx.plugin(LocalBashExecutor, { maxOutputBytes: 100, graceMs: 200 })
  246. ;(ctx.bash as LocalBashExecutor).internals = { spillDir }
  247. await ctx.plugin(ToolBash)
  248. const result = await call(ctx, 'bash', { command: 'for i in $(seq 1 100); do printf "line-%04d\\n" $i; done', description: 'test command' })
  249. expect(text(result)).toContain('[output truncated; full output: ')
  250. expect(text(result)).toContain('line-0100')
  251. })
  252. it('honors workdir', async () => {
  253. const ctx = await setup()
  254. const result = await call(ctx, 'bash', { command: 'pwd', description: 'test command', workdir: '/tmp' })
  255. expect(text(result).trim()).toMatch(/\/tmp$/)
  256. })
  257. it('surfaces spawn failures as isError', async () => {
  258. const ctx = await setup()
  259. const result = await call(ctx, 'bash', { command: 'true', description: 'test command', workdir: '/nonexistent-dsh' })
  260. expect(result.isError).toBe(true)
  261. expect(text(result)).toMatch(/ENOENT/)
  262. })
  263. it('surfaces foreground aborts as isError', async () => {
  264. const ctx = await setup()
  265. const controller = new AbortController()
  266. const pending = ctx.tools.execute({
  267. callId: CallId('call-abort'),
  268. name: 'bash',
  269. arguments: { command: 'sleep 60', description: 'test command' },
  270. signal: controller.signal,
  271. })
  272. setTimeout(() => { controller.abort() }, 50)
  273. const result = await pending
  274. expect(result.isError).toBe(true)
  275. expect(text(result)).toMatch(/aborted/)
  276. })
  277. // Type and required-key violations are rejected by the harness
  278. // (defineTool validates against the ParameterSchemaSpec — the arg-validation Agent Note) before execute.
  279. it.each([
  280. [{}, /missing required property "command"/],
  281. [{ command: 42, description: 'd' }, /"command" must be a string/],
  282. [{ command: 'x' }, /missing required property "description"/],
  283. [{ command: 'x', description: 7 }, /"description" must be a string/],
  284. [{ command: 'x', description: 'd', timeoutMs: 'soon' }, /"timeoutMs" must be a number/],
  285. [{ command: 'x', description: 'd', workdir: 7 }, /"workdir" must be a string/],
  286. [{ command: 'x', description: 'd', run_in_background: 'yes' }, /"run_in_background" must be a boolean/],
  287. ])('rejects schema-invalid args %j', async (args, pattern) => {
  288. const ctx = await setup()
  289. const result = await call(ctx, 'bash', args)
  290. expect(result.isError).toBe(true)
  291. expect(text(result)).toMatch(pattern)
  292. })
  293. // Value constraints the ParameterSchemaSpec can't express stay in the tool body.
  294. it.each([
  295. [{ command: ' ', description: 'd' }, /invalid command/],
  296. [{ command: 'x', description: ' ' }, /invalid description/],
  297. [{ command: 'x', description: 'd', timeoutMs: -1 }, /invalid timeoutMs/],
  298. ])('rejects value-invalid args %j', async (args, pattern) => {
  299. const ctx = await setup()
  300. const result = await call(ctx, 'bash', args)
  301. expect(result.isError).toBe(true)
  302. expect(text(result)).toMatch(pattern)
  303. })
  304. it('rejects a non-JSON numeric argument before tool-specific validation', async () => {
  305. const ctx = await setup()
  306. const result = await call(ctx, 'bash', {
  307. command: 'x', description: 'd', timeoutMs: Number.NaN,
  308. })
  309. expect(result.isError).toBe(true)
  310. expect(text(result)).toContain('tool execution arguments must be losslessly JSON-serializable')
  311. })
  312. it('registers the bash schema with run_in_background exposed by default', async () => {
  313. const ctx = await setup()
  314. const schemas = ctx.tools.schemas()
  315. expect(schemas.map(schema => schema.name)).toEqual(['bash'])
  316. const bashSchema = schemas[0]!
  317. expect(bashSchema.parameters).toMatchObject({
  318. type: 'object',
  319. required: ['command', 'description'],
  320. })
  321. expect(Object.keys(bashSchema.parameters.properties as Record<string, unknown>))
  322. .toContain('run_in_background')
  323. expect(bashSchema.description).toContain('task_output')
  324. })
  325. it('contributes the exit-code habit as its prompt section (guidance the descriptions cannot carry)', async () => {
  326. const ctx = await setup()
  327. ctx.systemPrompt.section({ name: 'test:before-bash', order: 104, text: 'before' })
  328. ctx.systemPrompt.section({ name: 'test:after-bash', order: 106, text: 'after' })
  329. const assembly = await ctx.systemPrompt.assemble()
  330. const section = assembly.sections.find(s => s.name === 'tool:bash')
  331. expect(assembly.sections.map(s => s.name)).toEqual([
  332. 'harness:identity',
  333. 'deployment:persona',
  334. 'test:before-bash',
  335. 'tool:bash',
  336. 'test:after-bash',
  337. ])
  338. expect(section?.text).toContain('[exit code: N]')
  339. })
  340. it('unregisters everything when the plugin fiber is disposed (HMR safety)', async () => {
  341. const ctx = new Context()
  342. await ctx.plugin(SystemPrompt)
  343. await ctx.plugin(ToolRegistry)
  344. await ctx.plugin(LocalBashExecutor, {})
  345. const fiber = await ctx.plugin(ToolBash)
  346. expect(ctx.tools.schemas()).toHaveLength(1)
  347. expect((await ctx.systemPrompt.assemble()).sections.map(s => s.name)).toEqual(['harness:identity', 'deployment:persona', 'tool:bash'])
  348. await fiber.dispose()
  349. expect(ctx.tools.schemas()).toHaveLength(0)
  350. // Only the system-prompt plugin's own built-in sections remain.
  351. expect((await ctx.systemPrompt.assemble()).sections.map(s => s.name)).toEqual(['harness:identity', 'deployment:persona'])
  352. })
  353. it('tools depend on the executor: no registration without ctx.bash', async () => {
  354. const ctx = new Context()
  355. await ctx.plugin(SystemPrompt)
  356. await ctx.plugin(ToolRegistry)
  357. // inject: ['tools', 'bash'] keeps the plugin pending until bash exists.
  358. await ctx.plugin(ToolBash)
  359. expect(ctx.tools.schemas()).toHaveLength(0)
  360. await ctx.plugin(LocalBashExecutor, {})
  361. await new Promise(resolve => setTimeout(resolve, 0))
  362. expect(ctx.tools.schemas()).toHaveLength(1)
  363. })
  364. it('applies the built-in background default when apply() receives a bare config', async () => {
  365. // Bypasses the schemastery defaults on purpose: apply() must stand on its
  366. // own `?? true` fallback when embedded programmatically without the schema.
  367. const ctx = new Context()
  368. await ctx.plugin(SystemPrompt)
  369. await ctx.plugin(ToolRegistry)
  370. await ctx.plugin(LocalBashExecutor, {})
  371. ToolBash.apply(ctx, {})
  372. const schema = ctx.tools.schemas()[0]!
  373. expect(Object.keys(schema.parameters.properties as Record<string, unknown>))
  374. .toContain('run_in_background')
  375. })
  376. })
  377. describe('background execution through the task runtime', () => {
  378. it('run_in_background acks with the task id, readable through the REAL task_output tool', async () => {
  379. const ctx = await setupWithTasks()
  380. const started = await call(ctx, 'bash', { command: 'echo bg-ok', description: 'test command', run_in_background: true })
  381. expect(started.isError).toBe(false)
  382. if (started.isError) throw new Error('expected background bash success')
  383. expect(started.value).toEqual({ kind: 'background', taskId: 'bash-1' })
  384. expect(text(started)).toBe('started background task bash-1')
  385. const read = await callUntilText(ctx, 'task_output', { task_id: 'bash-1' }, 'bg-ok')
  386. expect(text(read)).toContain('bg-ok')
  387. // A later read reports the terminal outcome in the generic status line.
  388. const final = await callUntilText(ctx, 'task_output', { task_id: 'bash-1' }, '[status: completed, exit code: 0]')
  389. expect(final.isError).toBe(false)
  390. })
  391. it('a running background task is killable through the REAL task_kill tool', async () => {
  392. const ctx = await setupWithTasks()
  393. await call(ctx, 'bash', { command: 'sleep 60', description: 'test command', run_in_background: true })
  394. const killed = await call(ctx, 'task_kill', { task_id: 'bash-1' })
  395. expect(text(killed)).toBe('requested cancellation of task bash-1')
  396. // The cancel reached the process handle; the task settles as killed with
  397. // the signal detail mapped by processOutcome.
  398. const final = await call(ctx, 'task_output', { task_id: 'bash-1', wait: true })
  399. expect(text(final)).toContain('[status: killed, signal: SIGTERM]')
  400. })
  401. it('a self-signal background exit is reported as killed through the REAL task_output tool', async () => {
  402. const ctx = await setupWithTasks()
  403. await call(ctx, 'bash', { command: 'kill -TERM $$', description: 'test command', run_in_background: true })
  404. const final = await call(ctx, 'task_output', { task_id: 'bash-1', wait: true })
  405. expect(text(final)).toContain('[status: killed, signal: SIGTERM]')
  406. })
  407. it('a background task started by an agent is registered with that agent as owner', async () => {
  408. // The producer must forward exec.agent as the task owner.
  409. const ctx = await setupWithTasks()
  410. const agent = registerFakeAgent(ctx, 'sess-owner')
  411. const started = await call(ctx, 'bash', { command: 'sleep 60', description: 'test command', run_in_background: true }, agent)
  412. expect(text(started)).toBe('started background task bash-1')
  413. const anon = await call(ctx, 'task_output', { task_id: 'bash-1' })
  414. expect(anon.isError).toBe(true)
  415. expect(text(anon)).toMatch(/belongs to another session/)
  416. const killed = await call(ctx, 'task_kill', { task_id: 'bash-1' }, agent)
  417. expect(killed.isError).toBe(false)
  418. await call(ctx, 'task_output', { task_id: 'bash-1', wait: true }, agent) // await settlement — no orphan
  419. })
  420. it('fails loud when the task runtime is not loaded', async () => {
  421. const ctx = await setup() // no TaskService / ToolTasks
  422. const result = await call(ctx, 'bash', { command: 'sleep 60', description: 'test command', run_in_background: true })
  423. expect(result.isError).toBe(true)
  424. expect(text(result)).toContain('background tasks unavailable: load @deepseek-ai/dsh-tasks and @deepseek-ai/dsh-tool-tasks')
  425. })
  426. it('a pre-aborted call refuses to start: isError, no process spawned', async () => {
  427. const ctx = new Context()
  428. await ctx.plugin(SystemPrompt)
  429. await ctx.plugin(ToolRegistry)
  430. await ctx.plugin(AgentRegistry)
  431. await ctx.plugin(TaskService)
  432. await ctx.plugin(ToolTasks)
  433. await ctx.plugin(CountingStartExecutor)
  434. await ctx.plugin(ToolBash)
  435. const controller = new AbortController()
  436. controller.abort()
  437. const result = await ctx.tools.execute({
  438. callId: CallId('call-pre-aborted'),
  439. name: 'bash',
  440. arguments: { command: 'sleep 60', description: 'test command', run_in_background: true },
  441. signal: controller.signal,
  442. })
  443. expect(result.isError).toBe(true)
  444. expect(text(result)).toContain('command aborted')
  445. expect((ctx.bash as CountingStartExecutor).starts).toBe(0)
  446. })
  447. it('never spawns the process when tasks.start preflight throws (no orphan, by construction)', async () => {
  448. // With no control surface, task preflight fails before the executor can spawn.
  449. const ctx = new Context()
  450. await ctx.plugin(SystemPrompt)
  451. await ctx.plugin(ToolRegistry)
  452. await ctx.plugin(AgentRegistry)
  453. await ctx.plugin(TaskService)
  454. await ctx.plugin(CountingStartExecutor)
  455. await ctx.plugin(ToolBash)
  456. const result = await call(ctx, 'bash', { command: 'sleep 60', description: 'test command', run_in_background: true })
  457. expect(result.isError).toBe(true)
  458. expect(text(result)).toContain('no control surface is attached')
  459. // Declare-then-execute: the failed preflight means no process ever ran.
  460. expect((ctx.bash as CountingStartExecutor).starts).toBe(0)
  461. })
  462. it('enableRunInBackground: false removes the parameter and flips the description', async () => {
  463. const ctx = new Context()
  464. await ctx.plugin(SystemPrompt)
  465. await ctx.plugin(ToolRegistry)
  466. await ctx.plugin(LocalBashExecutor, {})
  467. await ctx.plugin(ToolBash, { enableRunInBackground: false })
  468. const schema = ctx.tools.schemas().find(s => s.name === 'bash')!
  469. expect(Object.keys(schema.parameters.properties as Record<string, unknown>))
  470. .toEqual(['command', 'description', 'timeoutMs', 'workdir'])
  471. expect(schema.description).toContain('Background execution is not available')
  472. expect(schema.description).not.toContain('run_in_background')
  473. // The registry-held definition agrees (schema and capability never disagree).
  474. const parameters = ctx.tools.get('bash')!.parameters as { properties: Record<string, unknown> }
  475. expect('run_in_background' in parameters.properties).toBe(false)
  476. // Schema omission is advertising; execution must also enforce the opt-out.
  477. const forced = await call(ctx, 'bash', { command: 'echo hi', description: 'test command', run_in_background: true })
  478. expect(forced.isError).toBe(true)
  479. expect(text(forced)).toContain('run_in_background is disabled for this deployment')
  480. const foreground = await call(ctx, 'bash', { command: 'echo hi', description: 'test command' })
  481. expect(foreground.isError).toBe(false)
  482. })
  483. })
  484. describe('sandbox escalation through the generic task producer', () => {
  485. const escalate = {
  486. command: 'true',
  487. description: 'test escalation',
  488. sandbox_permissions: 'workspace-write',
  489. justification: 'the command needs workspace writes',
  490. }
  491. it('advertises the sandbox fields and validates their pairing', async () => {
  492. const { ctx } = await setupSandboxed()
  493. const schema = ctx.tools.schemas().find(item => item.name === 'bash')!
  494. const properties = schema.parameters.properties as Record<string, { enum?: string[] }>
  495. expect(properties['sandbox_permissions']?.enum).toEqual(['workspace-write', 'danger-full-access'])
  496. expect(schema.description).toContain('approval prompt')
  497. for (const args of [
  498. { command: 'true', description: 'd', sandbox_permissions: 'workspace-write' },
  499. { command: 'true', description: 'd', justification: 'why' },
  500. { command: 'true', description: 'd', sandbox_permissions: 'workspace-write', justification: ' ' },
  501. ]) {
  502. expect((await call(ctx, 'bash', args)).isError).toBe(true)
  503. }
  504. })
  505. it('rejects injected escalation without a sandbox and non-widening escalation without prompting', async () => {
  506. const plain = await setup()
  507. expect(text(await call(plain, 'bash', escalate))).toContain('not available in this composition')
  508. const { ctx } = await setupSandboxed(true)
  509. const prompted = vi.fn()
  510. ctx.on('approval/request', () => { prompted(); return Promise.resolve<ApprovalOutcome>('allowed-once') })
  511. const result = await call(ctx, 'bash', { ...escalate, sandbox_permissions: 'workspace-write' }, sandboxAgent('workspace-write'))
  512. expect(text(result)).toContain('not strictly wider')
  513. expect(prompted).not.toHaveBeenCalled()
  514. const malformed = sandboxAgent()
  515. ;(malformed.session.events as unknown as Array<{ type: string; data: { mode: string } }>).push({
  516. type: 'sandbox/mode',
  517. data: { mode: 'unknown-mode' },
  518. })
  519. expect(text(await call(ctx, 'bash', escalate, malformed))).toContain('not strictly wider')
  520. })
  521. it('fails closed when approval cannot be routed', async () => {
  522. const withoutService = await setupSandboxed()
  523. expect(text(await call(withoutService.ctx, 'bash', escalate, sandboxAgent()))).toContain('no approval service')
  524. const withService = await setupSandboxed(true)
  525. expect(text(await call(withService.ctx, 'bash', escalate))).toContain('no agent to route')
  526. expect(text(await call(withService.ctx, 'bash', escalate, sandboxAgent()))).toContain('no approval channel')
  527. })
  528. it.each([
  529. ['rejected', 'user rejected'],
  530. ['cancelled', 'was cancelled'],
  531. ] as const)('maps an approval %s to its distinct failure', async (outcome, message) => {
  532. const { ctx, bash } = await setupSandboxed(true)
  533. ctx.on('approval/request', () => Promise.resolve<ApprovalOutcome>(outcome))
  534. const result = await call(ctx, 'bash', escalate, sandboxAgent())
  535. expect(text(result)).toContain(message)
  536. expect(bash.modes).toEqual([])
  537. })
  538. it('runs a granted foreground or background call under the approved mode', async () => {
  539. const { ctx, bash } = await setupSandboxed(true)
  540. ctx.on('approval/request', () => Promise.resolve<ApprovalOutcome>('allowed-once'))
  541. const agent = sandboxAgent(undefined, ctx)
  542. ctx.agents.register(agent)
  543. const foreground = await ctx.tools.execute({
  544. callId: CallId('sandbox-signal'),
  545. name: 'bash',
  546. arguments: escalate,
  547. agent,
  548. signal: new AbortController().signal,
  549. })
  550. expect(foreground.isError).toBe(false)
  551. const background = await call(ctx, 'bash', { ...escalate, run_in_background: true }, agent)
  552. expect(text(background)).toBe('started background task bash-1')
  553. expect(bash.modes).toEqual(['workspace-write', 'workspace-write'])
  554. })
  555. it('uses the session override for ordinary calls and evaluates widening against it', async () => {
  556. const { ctx, bash } = await setupSandboxed(true)
  557. const agent = sandboxAgent('workspace-write')
  558. await call(ctx, 'bash', { command: 'true', description: 'ordinary' }, agent)
  559. ctx.on('approval/request', () => Promise.resolve<ApprovalOutcome>('allowed-once'))
  560. await call(ctx, 'bash', { ...escalate, sandbox_permissions: 'danger-full-access' }, agent)
  561. expect(bash.modes).toEqual(['workspace-write', 'danger-full-access'])
  562. })
  563. it('omits sandbox facts the executor did not acquire from the canonical result', async () => {
  564. const { ctx } = await setupSandboxed()
  565. const result = await call(ctx, 'bash', {
  566. command: 'without optional sandbox facts',
  567. description: 'exercise optional sandbox facts',
  568. })
  569. if (result.isError) throw new Error('expected foreground bash success')
  570. expect(result.value).toMatchObject({
  571. kind: 'foreground',
  572. sandbox: { mode: 'read-only', denied: false },
  573. })
  574. expect((result.value as { sandbox: object }).sandbox).not.toHaveProperty('enforcement')
  575. expect((result.value as { sandbox: object }).sandbox).not.toHaveProperty('runnerFailed')
  576. })
  577. it('keeps the exhaustiveness backstop for a rogue approval implementation', async () => {
  578. const { ctx } = await setupSandboxed(true)
  579. ctx.approval.request = () => Promise.resolve('rogue' as ApprovalOutcome)
  580. const result = await call(ctx, 'bash', escalate, sandboxAgent())
  581. expect(text(result)).toContain('unreachable variant in EscalationOutcome')
  582. })
  583. })
  584. describe('renderProcessRead', () => {
  585. const base: BashProcessRead = { delta: 'out\n', lossy: false }
  586. it('returns the delta verbatim for a lossless read', () => {
  587. expect(renderProcessRead(base)).toBe('out\n')
  588. expect(renderProcessRead({ delta: '', lossy: false })).toBe('')
  589. })
  590. it('appends the loss notice with the available spill paths', () => {
  591. expect(renderProcessRead({ ...base, lossy: true, stdoutSpillPath: '/spill/out.log' }))
  592. .toBe('out\n[some output was dropped from memory; full output: /spill/out.log]')
  593. expect(renderProcessRead({ ...base, lossy: true, stdoutSpillPath: '/spill/out.log', stderrSpillPath: '/spill/err.log' }))
  594. .toBe('out\n[some output was dropped from memory; full output: /spill/out.log, /spill/err.log]')
  595. })
  596. it('reports (unavailable) when a lossy read has no safe spill path', () => {
  597. expect(renderProcessRead({ ...base, lossy: true }))
  598. .toBe('out\n[some output was dropped from memory; full output: (unavailable)]')
  599. })
  600. it('an empty lossy delta is the notice alone', () => {
  601. expect(renderProcessRead({ delta: '', lossy: true, stderrSpillPath: '/spill/err.log' }))
  602. .toBe('[some output was dropped from memory; full output: /spill/err.log]')
  603. })
  604. it('inserts the separating newline only when the delta lacks one', () => {
  605. expect(renderProcessRead({ delta: 'tail', lossy: true }))
  606. .toBe('tail\n[some output was dropped from memory; full output: (unavailable)]')
  607. expect(renderProcessRead({ delta: 'tail\n', lossy: true }))
  608. .toBe('tail\n[some output was dropped from memory; full output: (unavailable)]')
  609. })
  610. it('appends settled sandbox denial and runner-failure facts', () => {
  611. expect(renderProcessRead(base, { mode: 'read-only', denied: true }, ['workspace-write']))
  612. .toContain('[sandbox: escalation available')
  613. expect(renderProcessRead({ delta: 'tail', lossy: false }, { mode: 'read-only', denied: true }))
  614. .toBe('tail\n[sandbox: file access denied under read-only mode]')
  615. const runner = renderProcessRead(
  616. { delta: '', lossy: false },
  617. { mode: 'workspace-write', denied: true, runnerFailed: true },
  618. ['danger-full-access'],
  619. )
  620. expect(runner).toContain('sandbox runner itself failed under workspace-write mode')
  621. expect(runner).not.toContain('file access denied')
  622. })
  623. })
  624. describe('processOutcome', () => {
  625. function settled(over: Partial<BashProcess>): BashProcess {
  626. return {
  627. status: 'completed',
  628. exitCode: 0,
  629. signal: null,
  630. done: Promise.resolve(),
  631. readOutput: () => ({ delta: '', lossy: false }),
  632. kill: () => false,
  633. ...over,
  634. }
  635. }
  636. it('maps a signal-killed process to killed with the signal detail', () => {
  637. expect(processOutcome(settled({ status: 'killed', signal: 'SIGTERM' })))
  638. .toEqual({ status: 'killed', detail: 'signal: SIGTERM' })
  639. })
  640. it('maps a killed process without a recorded signal (kill raced exit / spawn failure)', () => {
  641. expect(processOutcome(settled({ status: 'killed', exitCode: null })))
  642. .toEqual({ status: 'killed', detail: 'killed before exit' })
  643. })
  644. it('maps a completed process to its exit code', () => {
  645. expect(processOutcome(settled({ exitCode: 3 })))
  646. .toEqual({ status: 'completed', detail: 'exit code: 3' })
  647. })
  648. it('defensively reads a null exit code as 0 (handle shapes from other executors)', () => {
  649. expect(processOutcome(settled({ exitCode: null })))
  650. .toEqual({ status: 'completed', detail: 'exit code: 0' })
  651. })
  652. })
  653. describe('session-cwd routing (per-session workdir)', () => {
  654. // An agent whose session header carries a cwd (what session/new records).
  655. const agentInCwd = (cwd: string) =>
  656. ({ inject: () => undefined, session: { header: { version: 0, id: 'c', createdAt: 0, cwd } } }) as unknown as Agent
  657. it('defaults bash to the agent\'s session cwd (not the server launch dir)', async () => {
  658. const ctx = await setup()
  659. const result = await call(ctx, 'bash', { command: 'pwd', description: 'pwd' }, agentInCwd('/tmp'))
  660. expect(text(result).trim()).toMatch(/\/tmp$/)
  661. })
  662. it('an explicit absolute workdir overrides the session cwd', async () => {
  663. const ctx = await setup()
  664. const result = await call(ctx, 'bash', { command: 'pwd', description: 'pwd', workdir: '/tmp' }, agentInCwd('/'))
  665. expect(text(result).trim()).toMatch(/\/tmp$/)
  666. })
  667. it('a relative workdir is resolved against the session cwd', async () => {
  668. const ctx = await setup()
  669. // session cwd /usr + relative 'bin' → /usr/bin
  670. const result = await call(ctx, 'bash', { command: 'pwd', description: 'pwd', workdir: 'bin' }, agentInCwd('/usr'))
  671. expect(text(result).trim()).toMatch(/\/usr\/bin$/)
  672. })
  673. it('two sessions with different cwds each run bash in their own dir', async () => {
  674. const ctx = await setup()
  675. const inUsr = await call(ctx, 'bash', { command: 'pwd', description: 'pwd' }, agentInCwd('/usr'))
  676. const inTmp = await call(ctx, 'bash', { command: 'pwd', description: 'pwd' }, agentInCwd('/tmp'))
  677. expect(text(inUsr).trim()).toMatch(/\/usr$/)
  678. expect(text(inTmp).trim()).toMatch(/\/tmp$/)
  679. })
  680. it('falls back to the executor default when the agent has no session cwd', async () => {
  681. const ctx = await setup()
  682. // No exec.agent at all → executor uses its config/process.cwd() default.
  683. const result = await ctx.tools.execute({ callId: CallId('cwd-noagent'), name: 'bash', arguments: { command: 'pwd', description: 'pwd' } })
  684. expect(result.isError).toBe(false)
  685. expect(text(result).trim().length).toBeGreaterThan(0)
  686. })
  687. })
  688. describe('renderResult', () => {
  689. const base = {
  690. exitCode: 0 as number | null,
  691. signal: null as NodeJS.Signals | null,
  692. timedOut: false,
  693. aborted: false,
  694. timeoutMs: 1000,
  695. stdout: { text: '', truncated: false },
  696. stderr: { text: '', truncated: false },
  697. }
  698. it('renders stderr-only output without a stdout prefix', () => {
  699. expect(renderResult({ ...base, stderr: { text: 'err\n', truncated: false } }))
  700. .toBe('[stderr]\nerr\n')
  701. })
  702. it('adds a separator when stdout does not end with a newline', () => {
  703. expect(renderResult({
  704. ...base,
  705. stdout: { text: 'out', truncated: false },
  706. stderr: { text: 'err', truncated: false },
  707. })).toBe('out\n[stderr]\nerr')
  708. })
  709. it('appends exit-code markers after a newline for unterminated output', () => {
  710. expect(renderResult({ ...base, exitCode: 7, stdout: { text: 'x', truncated: false } }))
  711. .toBe('x\n[exit code: 7]')
  712. })
  713. it('renders signal kills without the timeout marker when not timed out', () => {
  714. expect(renderResult({ ...base, exitCode: null, signal: 'SIGKILL' }))
  715. .toBe('(no output)\n[killed by signal: SIGKILL]')
  716. })
  717. it('reports a timeout that exited 0 (trapped signal) without a kill marker', () => {
  718. expect(renderResult({ ...base, exitCode: 0, signal: null, timedOut: true }))
  719. .toBe('(no output)\n[timed out after 1000ms]')
  720. })
  721. it('orders the timeout marker before a kill marker', () => {
  722. expect(renderResult({ ...base, exitCode: null, signal: 'SIGTERM', timedOut: true }))
  723. .toBe('(no output)\n[timed out after 1000ms]\n[killed by signal: SIGTERM]')
  724. })
  725. it('notes truncation with a fallback when the spill path is missing', () => {
  726. expect(renderResult({ ...base, stdout: { text: 'tail', truncated: true } }))
  727. .toBe('tail\n[output truncated; full output: (unavailable)]')
  728. })
  729. it('reports sandbox denials before exit status and hints only when escalation is advertised', () => {
  730. const result: BashRunResult = {
  731. exitCode: 1,
  732. signal: null,
  733. timedOut: false,
  734. aborted: false,
  735. timeoutMs: 1000,
  736. stdout: { text: '', truncated: false },
  737. stderr: { text: 'denied', truncated: false },
  738. sandbox: { mode: 'read-only', denied: true },
  739. }
  740. expect(renderResult(result)).toMatch(/denied under read-only mode\]\n\[exit code: 1\]$/)
  741. expect(renderResult(result, ['workspace-write'])).toContain('[sandbox: escalation available')
  742. expect(renderResult({ ...result, sandbox: { mode: 'read-only', denied: false } }, ['workspace-write']))
  743. .not.toContain('[sandbox:')
  744. })
  745. })
  746. describe('tool-owned UI presentation (presentCall / presentResult)', () => {
  747. it('bash presentCall: a foreground run is a terminal card (command title, description, workdir → cwd absolute or relative)', async () => {
  748. const ctx = await setup()
  749. // No explicit workdir → a terminal card with no cwd (the UI bridge fills the
  750. // session cwd it owns; the pure presenter can't see it).
  751. expect(ctx.tools.get('bash')?.presentCall?.({ command: 'ls -la src', description: 'List files in src' }))
  752. .toEqual({ card: 'terminal', title: 'ls -la src', description: 'List files in src' })
  753. // An ABSOLUTE workdir is surfaced verbatim as the terminal cwd header.
  754. expect(ctx.tools.get('bash')?.presentCall?.({ command: 'pwd', description: 'Print dir', workdir: '/tmp/x' }))
  755. .toEqual({ card: 'terminal', title: 'pwd', description: 'Print dir', cwd: '/tmp/x' })
  756. // A RELATIVE workdir is passed through AS-IS (the bridge resolves it against
  757. // the session cwd, matching where execution runs) — not dropped.
  758. expect(ctx.tools.get('bash')?.presentCall?.({ command: 'pwd', description: 'Print dir', workdir: 'sub' }))
  759. .toEqual({ card: 'terminal', title: 'pwd', description: 'Print dir', cwd: 'sub' })
  760. })
  761. it('bash presentResult: a terminal result carries RAW output (newlines intact) + parsed exit code', async () => {
  762. const ctx = await setup()
  763. const present = ctx.tools.get('bash')!.presentResult!(
  764. { command: 'echo hi', description: 'echo' },
  765. { content: [{ type: 'text', text: 'hi\n[exit code: 0]\n\n' }], isError: false },
  766. )
  767. // A terminal result keeps the RAW bytes (newlines intact) a terminal renderer
  768. // needs; the bridge derives the fenced fallback. exitCode is parsed back from
  769. // the [exit code: N] marker.
  770. expect(present).toEqual({ card: 'terminal', output: 'hi\n[exit code: 0]\n\n', exitCode: 0 })
  771. })
  772. it('bash presentResult: a non-zero exit and a signal kill parse into exitCode / signal', async () => {
  773. const ctx = await setup()
  774. const args = { command: 'x', description: 'x' }
  775. const nonzero = ctx.tools.get('bash')!.presentResult!(args, { content: [{ type: 'text', text: 'oops\n[exit code: 3]' }], isError: false })
  776. expect(nonzero).toEqual({ card: 'terminal', output: 'oops\n[exit code: 3]', exitCode: 3 })
  777. const killed = ctx.tools.get('bash')!.presentResult!(args, { content: [{ type: 'text', text: 'gone\n[killed by signal: SIGKILL]' }], isError: false })
  778. expect(killed).toEqual({ card: 'terminal', output: 'gone\n[killed by signal: SIGKILL]', signal: 'SIGKILL' })
  779. })
  780. it('bash presentResult exit parse is the inverse of renderResult markers (round-trip)', async () => {
  781. const ctx = await setup()
  782. const present = ctx.tools.get('bash')!
  783. // For each renderResult outcome, the rendered text fed back through
  784. // presentResult recovers the matching structured exit — the parse and the
  785. // marker emission co-evolve in one file, so this pins the pair.
  786. const base = {
  787. aborted: false,
  788. timeoutMs: 1000,
  789. stdout: { text: 'out', truncated: false },
  790. stderr: { text: '', truncated: false },
  791. }
  792. const cases = [
  793. { result: { ...base, exitCode: 0, signal: null, timedOut: false }, expect: { exitCode: 0 } },
  794. { result: { ...base, exitCode: 7, signal: null, timedOut: false }, expect: { exitCode: 7 } },
  795. { result: { ...base, exitCode: null, signal: 'SIGTERM' as const, timedOut: false }, expect: { signal: 'SIGTERM' } },
  796. // A trapped-timeout run that exits 0 has no signal/exit marker → reads as exit 0 (it did exit 0).
  797. { result: { ...base, exitCode: 0, signal: null, timedOut: true }, expect: { exitCode: 0 } },
  798. ]
  799. for (const c of cases) {
  800. const rendered = renderResult(c.result)
  801. const out = present.presentResult!({ command: 'x', description: 'x' }, { content: [{ type: 'text', text: rendered }], isError: false })
  802. // Drop card + output; the remaining fields are the parsed exit.
  803. const { card: _c, output: _o, ...exit } = out as { card: string; output?: string; exitCode?: number; signal?: string }
  804. expect(exit).toEqual(c.expect)
  805. }
  806. })
  807. it('bash presentResult: a clean exit-0 whose output ENDS in marker-like text is NOT read as a failure', async () => {
  808. const ctx = await setup()
  809. const args = { command: 'printf "[exit code: 5]"', description: 'print' }
  810. // A successful command may print marker-like text. A clean result appends no marker or
  811. // newline; parsing requires the leading newline emitted for real markers, so this stays exit 0.
  812. const out = ctx.tools.get('bash')!.presentResult!(args, { content: [{ type: 'text', text: '[exit code: 5]' }], isError: false })
  813. expect(out).toEqual({ card: 'terminal', output: '[exit code: 5]', exitCode: 0 })
  814. // Same for a fake signal marker with no leading newline.
  815. const sig = ctx.tools.get('bash')!.presentResult!(args, { content: [{ type: 'text', text: '[killed by signal: SIGKILL]' }], isError: false })
  816. expect(sig).toEqual({ card: 'terminal', output: '[killed by signal: SIGKILL]', exitCode: 0 })
  817. })
  818. it('bash presentCall/presentResult: a run_in_background call is a generic card and its ack carries no exit pill', async () => {
  819. const ctx = await setup()
  820. // The background start returns a task-id ack, not a streamed run — a generic
  821. // execute card with the command as rawInput and the description as content.
  822. const call = ctx.tools.get('bash')!.presentCall!({ command: 'sleep 100', description: 'wait', run_in_background: true })
  823. expect(call).toEqual({ card: 'generic', title: 'sleep 100', kind: 'execute', rawInput: 'sleep 100', content: [{ type: 'text', text: 'wait' }] })
  824. // The ack result is a generic fenced-text card — no terminal output / exit pill.
  825. const result = ctx.tools.get('bash')!.presentResult!(
  826. { command: 'sleep 100', description: 'wait', run_in_background: true },
  827. { content: [{ type: 'text', text: 'started background task bash-1' }], isError: false },
  828. )
  829. expect(result).toEqual({ card: 'generic', content: [{ type: 'text', text: '```console\nstarted background task bash-1\n```' }] })
  830. })
  831. it('bash presentResult: an isError result is a generic card (no real process exit to report)', async () => {
  832. const ctx = await setup()
  833. // A spawn failure / abort has no process exit — the body is an error message,
  834. // not renderResult output, so a generic fenced card, no terminal output/exit.
  835. const out = ctx.tools.get('bash')!.presentResult!(
  836. { command: 'x', description: 'x' },
  837. { content: [{ type: 'text', text: 'command aborted' }], isError: true },
  838. )
  839. expect(out).toEqual({ card: 'generic', content: [{ type: 'text', text: '```console\ncommand aborted\n```' }] })
  840. })
  841. it('bash presentResult: leaves a non-text (unexpected) result untouched → undefined (UI keeps raw content)', async () => {
  842. const ctx = await setup()
  843. const present = ctx.tools.get('bash')!.presentResult!(
  844. { command: 'x', description: 'x' },
  845. { content: [{ type: 'reasoning', text: 'unexpected' }], isError: false },
  846. )
  847. expect(present).toBeUndefined()
  848. })
  849. it('bash presentResult: a result that is not exactly one block → undefined (no single text to fence)', async () => {
  850. const ctx = await setup()
  851. const args = { command: 'x', description: 'x' }
  852. // Empty content (no block) and multi-block content both fall through.
  853. expect(ctx.tools.get('bash')!.presentResult!(args, { content: [], isError: false })).toBeUndefined()
  854. expect(ctx.tools.get('bash')!.presentResult!(args, {
  855. content: [{ type: 'text', text: 'a' }, { type: 'text', text: 'b' }],
  856. isError: false,
  857. })).toBeUndefined()
  858. })
  859. it('presentCall validates softly: malformed args (missing required description) return undefined, never throw', async () => {
  860. const ctx = await setup()
  861. // `defineTool` soft-validates replayed logged args before presentation. Invalid shapes return
  862. // undefined for generic UI rendering rather than throwing; `presentCall` accepts `unknown`.
  863. expect(ctx.tools.get('bash')?.presentCall?.({ command: 'ls' })).toBeUndefined()
  864. })
  865. })
  866. describe('the model-facing bash tool builds its request from named args only (no {...args} forward)', () => {
  867. const recordingDshHome = join(spillDir, 'dsh-home')
  868. /**
  869. * Records every {@link BashExecRequest} the consumer hands to `resolve()`, so a
  870. * test can assert what the model-facing tool DID and DID NOT forward. The `bash`
  871. * tool does not expose trusted-plugin fields (`stdoutMaxBytes`, `stdin`, or
  872. * `env`) as parameters, so it must build its request from named args only and
  873. * never spread unknown tool-call keys into it. This guard's job is to catch a
  874. * future refactor that blindly forwards `...args` — which would silently thread
  875. * model input into the post-scrub `env` merge or per-run capture budget — NOT
  876. * to defend a trust boundary
  877. * (the credential scrub in dsh-bash-local is the security control; see the
  878. * bash-stdin-env Agent Note). Foreground `run()` returns a canned result; `start()`
  879. * hands back an already-settled fake handle so the task registration completes.
  880. */
  881. class RecordingBashExecutor extends BashExecutor {
  882. readonly requests: BashExecRequest[] = []
  883. resolve(request: BashExecRequest): BashExecSpec {
  884. this.requests.push(request)
  885. return {
  886. command: request.command,
  887. workdir: request.workdir ?? process.cwd(),
  888. timeoutMs: request.timeoutMs ?? 0,
  889. stdoutMaxBytes: request.stdoutMaxBytes ?? 64_000,
  890. ...request.signal ? { signal: request.signal } : {},
  891. ...request.stdin !== undefined ? { stdin: request.stdin } : {},
  892. ...request.env !== undefined ? { env: request.env } : {},
  893. ...request.dshEnv !== undefined ? { dshEnv: request.dshEnv } : {},
  894. sandboxMode: request.sandboxMode,
  895. }
  896. }
  897. run(): Promise<BashRunResult> {
  898. return Promise.resolve({
  899. exitCode: 0, signal: null, timedOut: false, aborted: false, timeoutMs: 0,
  900. stdout: { text: 'ok', truncated: false }, stderr: { text: '', truncated: false },
  901. })
  902. }
  903. start(): BashProcess {
  904. return {
  905. status: 'completed',
  906. exitCode: 0,
  907. signal: null,
  908. done: Promise.resolve(),
  909. readOutput: () => ({ delta: '', lossy: false }),
  910. kill: () => false,
  911. }
  912. }
  913. }
  914. async function setupRecording(withJsonl = false) {
  915. const ctx = new Context()
  916. await ctx.plugin(SystemPrompt)
  917. await ctx.plugin(ToolRegistry)
  918. await ctx.plugin(AgentRegistry)
  919. if (withJsonl) {
  920. await ctx.plugin(SessionStore)
  921. await ctx.plugin(SessionPersistenceJsonl, { root: join(spillDir, 'jsonl') })
  922. }
  923. await ctx.plugin(TaskService)
  924. await ctx.plugin(ToolTasks)
  925. await ctx.plugin(RecordingBashExecutor)
  926. await ctx.plugin(ToolBash, { dshHome: recordingDshHome })
  927. return { ctx, bash: ctx.bash as RecordingBashExecutor }
  928. }
  929. it('describes the managed harness environment namespace to the model', async () => {
  930. const { ctx } = await setupRecording()
  931. const description = ctx.tools.get('bash')?.description ?? ''
  932. expect(description).toContain('$DSH_*')
  933. expect(description).not.toContain('DSH_SESSION_JSONL')
  934. })
  935. it('injects the session id and JSONL target path into a foreground request', async () => {
  936. const { ctx, bash } = await setupRecording(true)
  937. const agent = registerFakeAgent(ctx, 'request-fg', () => undefined)
  938. const path = ctx.sessionPersistence.locate(agent.session.header)?.path
  939. await ctx.tools.execute({
  940. callId: CallId('session-env-fg'),
  941. name: 'bash',
  942. arguments: { command: 'true', description: 'run command' },
  943. agent,
  944. })
  945. expect(bash.requests[0]?.dshEnv).toEqual({
  946. DSH_HOME: recordingDshHome,
  947. DSH_SESSION_ID: 'request-fg',
  948. DSH_SESSION_JSONL: path,
  949. DSH_SHELL: '1',
  950. })
  951. })
  952. it('injects the same trusted variables into a background request without forwarding model env', async () => {
  953. const { ctx, bash } = await setupRecording(true)
  954. const agent = registerFakeAgent(ctx, 'request-bg', () => undefined)
  955. const path = ctx.sessionPersistence.locate(agent.session.header)?.path
  956. await ctx.tools.execute({
  957. callId: CallId('session-env-bg'),
  958. name: 'bash',
  959. arguments: {
  960. command: 'sleep 1',
  961. description: 'run command',
  962. run_in_background: true,
  963. env: { DSH_SESSION_ID: 'spoofed', DSH_SESSION_JSONL: '/tmp/spoofed' },
  964. },
  965. agent,
  966. })
  967. expect(bash.requests[0]?.env).toBeUndefined()
  968. expect(bash.requests[0]?.dshEnv).toEqual({
  969. DSH_HOME: recordingDshHome,
  970. DSH_SESSION_ID: 'request-bg',
  971. DSH_SESSION_JSONL: path,
  972. DSH_SHELL: '1',
  973. })
  974. })
  975. it('injects built-ins and the stable session id when no JSONL locator is available', async () => {
  976. const { ctx, bash } = await setupRecording()
  977. const agent = registerFakeAgent(ctx, 'request-id-only', () => undefined)
  978. const ambient = process.env.DSH_SESSION_ID
  979. await ctx.tools.execute({
  980. callId: CallId('session-env-id-only'),
  981. name: 'bash',
  982. arguments: { command: 'true', description: 'run command' },
  983. agent,
  984. })
  985. expect(bash.requests[0]?.dshEnv).toEqual({
  986. DSH_HOME: recordingDshHome,
  987. DSH_SESSION_ID: 'request-id-only',
  988. DSH_SHELL: '1',
  989. })
  990. expect(process.env.DSH_SESSION_ID).toBe(ambient)
  991. })
  992. it('keeps parent and child agent session environments isolated', async () => {
  993. const { ctx, bash } = await setupRecording(true)
  994. const parent = registerFakeAgent(ctx, 'request-parent', () => undefined)
  995. const child = registerFakeAgent(ctx, 'request-child', () => undefined)
  996. for (const [callId, agent] of [['parent', parent], ['child', child]] as const) {
  997. await ctx.tools.execute({
  998. callId: CallId(`session-env-${callId}`),
  999. name: 'bash',
  1000. arguments: { command: 'true', description: 'run command' },
  1001. agent,
  1002. })
  1003. }
  1004. expect(bash.requests.map(request => request.dshEnv)).toEqual([
  1005. {
  1006. DSH_HOME: recordingDshHome,
  1007. DSH_SESSION_ID: 'request-parent',
  1008. DSH_SESSION_JSONL: ctx.sessionPersistence.locate(parent.session.header)?.path,
  1009. DSH_SHELL: '1',
  1010. },
  1011. {
  1012. DSH_HOME: recordingDshHome,
  1013. DSH_SESSION_ID: 'request-child',
  1014. DSH_SESSION_JSONL: ctx.sessionPersistence.locate(child.session.header)?.path,
  1015. DSH_SHELL: '1',
  1016. },
  1017. ])
  1018. expect(bash.requests[0]?.dshEnv?.DSH_SESSION_JSONL).not.toBe(bash.requests[1]?.dshEnv?.DSH_SESSION_JSONL)
  1019. })
  1020. it('does not forward trusted-only fields even when the model includes them as extra arguments', async () => {
  1021. const { ctx, bash } = await setupRecording()
  1022. // Unknown `env` and `stdin` keys are ignored by the schema and named request construction.
  1023. // This preserves the request shape; it is not a security boundary because shell syntax can
  1024. // already set environment variables or feed stdin.
  1025. await ctx.tools.execute({
  1026. callId: CallId('no-forward-1'),
  1027. name: 'bash',
  1028. arguments: {
  1029. command: 'echo hi',
  1030. description: 'echo',
  1031. env: { SNEAKY_API_KEY: 'leak' },
  1032. stdin: 'malicious payload',
  1033. stdoutMaxBytes: 999_999,
  1034. },
  1035. })
  1036. expect(bash.requests).toHaveLength(1)
  1037. const request = bash.requests[0]!
  1038. expect(request.command).toBe('echo hi')
  1039. expect('env' in request).toBe(false)
  1040. expect('stdin' in request).toBe(false)
  1041. expect('stdoutMaxBytes' in request).toBe(false)
  1042. })
  1043. it('a background bash call likewise carries no trusted-only fields', async () => {
  1044. const { ctx, bash } = await setupRecording()
  1045. const result = await ctx.tools.execute({
  1046. callId: CallId('no-forward-2'),
  1047. name: 'bash',
  1048. arguments: {
  1049. command: 'sleep 1',
  1050. description: 'sleep',
  1051. run_in_background: true,
  1052. env: { TOKEN: 'leak' },
  1053. stdin: 'x',
  1054. stdoutMaxBytes: 999_999,
  1055. },
  1056. })
  1057. // The call really went down the background path (the recorder sees the real
  1058. // request the consumer built, so the absent env/stdin below is a real
  1059. // negative, not a recorder that drops everything).
  1060. expect(text(result)).toBe('started background task bash-1')
  1061. expect(bash.requests).toHaveLength(1)
  1062. const request = bash.requests[0]!
  1063. expect(request.command).toBe('sleep 1')
  1064. expect('env' in request).toBe(false)
  1065. expect('stdin' in request).toBe(false)
  1066. expect('stdoutMaxBytes' in request).toBe(false)
  1067. })
  1068. })