sandbox.spec.ts 29 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632
  1. /**
  2. * Consumer-side `SandboxBashExecutor` tests. A fake Cordis sandbox service makes wrapping,
  3. * policy hand-off, fail-closed propagation, classification, and fact stamping deterministic;
  4. * real-provider integration lives in `tests/landlock.e2e.ts`. A mode-0555 directory supplies
  5. * the Unix denial signature used by the classifier without requiring a real sandbox runner.
  6. */
  7. import { chmodSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'
  8. import { tmpdir } from 'node:os'
  9. import { join, resolve } from 'node:path'
  10. import { afterAll, describe, expect, it, vi } from 'vitest'
  11. import { Context } from '@deepseek-ai/cordis'
  12. import type { ShellRunResult, CollectedOutput } from '@deepseek-ai/dsh-shell'
  13. import SessionProjectionRegistry from '@deepseek-ai/dsh-session-projection'
  14. import { SANDBOX_UNAVAILABLE, SandboxProvider, SandboxUnavailableError } from '@deepseek-ai/dsh-sandbox'
  15. import type { ConfinedArgv, SandboxExecutionPolicy, SandboxMode, SandboxPolicy } from '@deepseek-ai/dsh-sandbox'
  16. import { SandboxPolicyService } from '@deepseek-ai/dsh-sandbox-policy'
  17. import { SandboxBashExecutor } from '@deepseek-ai/dsh-bash-sandbox'
  18. import LocalSubprocessRuntime from '@deepseek-ai/dsh-subprocess-local'
  19. import type { SubprocessHandle, SubprocessOutputReader } from '@deepseek-ai/dsh-subprocess'
  20. import { classifyDenial } from '../src/helpers.ts'
  21. import type { Config } from '@deepseek-ai/dsh-bash-sandbox'
  22. const spillDir = mkdtempSync(join(tmpdir(), 'dsh-bash-sandbox-spec-'))
  23. afterAll(() => {
  24. rmSync(spillDir, { recursive: true, force: true })
  25. })
  26. /** One recorded provider call: the argv handed over and the policy it rode with. */
  27. interface ConfineCall {
  28. argv: string[]
  29. policy: SandboxPolicy
  30. }
  31. /** The Linux file-denial dialects the fake wraps carry — matches the unix-permission denials the tests below produce. */
  32. const UNIX_SIGNATURES = ['read-only file system', 'permission denied'] as const
  33. /** The runner-failure rule the fake wraps carry (a fake-runner: error line marks the sandbox itself failing). */
  34. const RUNNER_FAILURE = [{ fatalSignatures: ['fake-runner: '] }] as const
  35. /** Provider argv[0] forms that all share the caller-owned cwd spawn precondition. */
  36. const RUNNER_FORMS = [
  37. ['absolute', process.execPath],
  38. ['bare', 'node'],
  39. ['relative', './sandbox-runner'],
  40. ] as const
  41. /** A passthrough wrap: the caller's argv unchanged, asserted full — commands run unconfined, deterministically. */
  42. const passthrough = (argv: readonly string[]): ConfinedArgv =>
  43. ({ argv: [...argv], enforcement: 'full', denialSignatures: UNIX_SIGNATURES, runnerFailureRules: RUNNER_FAILURE })
  44. /**
  45. * Boot a context with a recording fake `ctx.sandbox` (behavior injectable
  46. * per test) and the executor under test on top of it.
  47. */
  48. async function setup(
  49. config: { mode?: SandboxMode; workspaceRoot?: string } & Config = {},
  50. behavior: (argv: readonly string[], policy: SandboxPolicy, signal?: AbortSignal) => ConfinedArgv | Promise<ConfinedArgv> = passthrough,
  51. ) {
  52. const { mode, workspaceRoot, ...execConfig } = config
  53. const calls: ConfineCall[] = []
  54. class FakeSandboxProvider extends SandboxProvider {
  55. async confine(argv: readonly string[], policy: SandboxPolicy, signal?: AbortSignal): Promise<ConfinedArgv> {
  56. calls.push({ argv: [...argv], policy })
  57. return behavior(argv, policy, signal)
  58. }
  59. }
  60. const ctx = new Context()
  61. await ctx.plugin(SessionProjectionRegistry)
  62. await ctx.plugin(FakeSandboxProvider)
  63. await ctx.plugin(SandboxPolicyService, {
  64. ...mode !== undefined ? { mode } : {},
  65. ...workspaceRoot !== undefined ? { workspaceRoot } : {},
  66. })
  67. await ctx.plugin(LocalSubprocessRuntime)
  68. ;(ctx.subprocess as LocalSubprocessRuntime).internals = { spillDir }
  69. await ctx.plugin(SandboxBashExecutor, { graceMs: 200, ...execConfig })
  70. const bash = ctx.shell as SandboxBashExecutor
  71. return { ctx, bash, calls }
  72. }
  73. function output(text: string): CollectedOutput {
  74. return { text, truncated: false }
  75. }
  76. function runResult(exitCode: number | null, stderr: string): ShellRunResult {
  77. return { exitCode, signal: null, timedOut: false, aborted: false, timeoutMs: 1000, stdout: output(''), stderr: output(stderr) }
  78. }
  79. function executionPolicy(mode: SandboxMode, workspaceRoot = resolve(process.cwd())): SandboxExecutionPolicy {
  80. return { mode, workspaceRoot }
  81. }
  82. describe('the provider hand-off', () => {
  83. it('preserves cancellation when a pending subprocess launch later rejects', async () => {
  84. const { ctx, bash } = await setup({}, () => passthrough(['node']))
  85. const entered = Promise.withResolvers<undefined>()
  86. const completion = Promise.withResolvers<never>()
  87. const controller = new AbortController()
  88. const reason = new Error('caller cancelled pending launch')
  89. const reader: SubprocessOutputReader = { readFrom: () => ({ text: '', nextOffset: 0, lossy: false }) }
  90. vi.spyOn(ctx.subprocess, 'spawn').mockImplementation(() => {
  91. entered.resolve(undefined)
  92. return {
  93. stdin: undefined, stdout: undefined, stderr: undefined, control: undefined,
  94. collected: { stdout: reader, stderr: reader }, done: completion.promise,
  95. terminate: () => {}, waitForExit: () => Promise.resolve(true),
  96. }
  97. })
  98. const pending = bash.run(bash.resolve({ command: 'true', signal: controller.signal }))
  99. const rejected = expect(pending).rejects.toBe(reason)
  100. try {
  101. await entered.promise
  102. controller.abort(reason)
  103. completion.reject(Object.assign(new Error('launch refused'), { code: 'ENOENT', syscall: 'spawn node', path: 'node' }))
  104. await rejected
  105. } finally {
  106. completion.reject(reason)
  107. await pending.catch(() => {})
  108. await ctx.fiber.dispose()
  109. }
  110. })
  111. it.each(['run', 'start'] as const)('cancels %s while confinement is pending without spawning', async (operation) => {
  112. const entered = Promise.withResolvers<AbortSignal>()
  113. const response = Promise.withResolvers<ConfinedArgv>()
  114. const { ctx, bash } = await setup({}, (_argv, _policy, signal) => {
  115. entered.resolve(signal!)
  116. return response.promise
  117. })
  118. const spawn = vi.spyOn(ctx.subprocess, 'spawn')
  119. const controller = new AbortController()
  120. const reason = new Error('cancel pending confinement')
  121. const pending = bash[operation](bash.resolve({ command: 'true', signal: controller.signal }))
  122. const rejected = expect(pending).rejects.toBe(reason)
  123. try {
  124. const signal = await entered.promise
  125. expect(spawn).not.toHaveBeenCalled()
  126. controller.abort(reason)
  127. expect(signal.aborted).toBe(true)
  128. response.resolve(passthrough(['bash', '-c', 'true']))
  129. await rejected
  130. expect(spawn).not.toHaveBeenCalled()
  131. } finally {
  132. response.resolve(passthrough(['bash', '-c', 'true']))
  133. await pending.catch(() => {})
  134. await ctx.fiber.dispose()
  135. }
  136. })
  137. it('publishes a background process only after confinement completes', async () => {
  138. const entered = Promise.withResolvers<undefined>()
  139. const response = Promise.withResolvers<ConfinedArgv>()
  140. const { ctx, bash } = await setup({}, () => { entered.resolve(undefined); return response.promise })
  141. const spawn = vi.spyOn(ctx.subprocess, 'spawn')
  142. let published = false
  143. const pending = bash.start(bash.resolve({ command: 'printf ready' })).then((process) => { published = true; return process })
  144. try {
  145. await entered.promise
  146. expect(published).toBe(false)
  147. expect(spawn).not.toHaveBeenCalled()
  148. response.resolve(passthrough(['bash', '-c', 'printf ready']))
  149. const process = await pending
  150. await process.done
  151. expect(process.readOutput().delta).toBe('ready')
  152. expect(spawn).toHaveBeenCalledOnce()
  153. } finally {
  154. response.resolve(passthrough(['bash', '-c', 'printf ready']))
  155. await ctx.fiber.dispose()
  156. }
  157. })
  158. it('hands the provider the exact bash argv and the per-call policy, and runs the returned argv', async () => {
  159. const { bash, calls } = await setup()
  160. const result = await bash.run(bash.resolve({ command: 'echo \'a b\' "c\'d"' }))
  161. expect(result.stdout.text).toBe('a b c\'d\n')
  162. expect(result.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full' })
  163. expect(calls).toEqual([{
  164. argv: ['bash', '-c', 'echo \'a b\' "c\'d"'],
  165. policy: { mode: 'read-only', workspaceRoot: resolve(process.cwd()) },
  166. }])
  167. })
  168. it('hands the provider\'s returned argv directly to ctx.subprocess.spawn', async () => {
  169. const returnedArgv = ['env', 'DSH_WRAP=1', 'bash', '-c', 'printf "%s" "$DSH_WRAP"']
  170. const { ctx, bash } = await setup({}, () => ({ argv: returnedArgv, enforcement: 'full', denialSignatures: UNIX_SIGNATURES, runnerFailureRules: RUNNER_FAILURE }))
  171. const spawn = vi.spyOn(ctx.subprocess, 'spawn')
  172. const result = await bash.run(bash.resolve({ command: 'printf "%s" "$DSH_WRAP"' }))
  173. expect(result.stdout.text).toBe('1')
  174. expect(spawn).toHaveBeenCalledTimes(1)
  175. expect(spawn.mock.calls[0]?.[0].argv).toEqual(returnedArgv)
  176. expect(result.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full' })
  177. })
  178. it('starts a non-Bash runner before the confined inner Bash evaluates BASH_ENV', async () => {
  179. const dir = mkdtempSync(join(tmpdir(), 'dsh-bash-env-order-'))
  180. const hook = join(dir, 'hook.sh')
  181. const order = join(dir, 'order.txt')
  182. writeFileSync(hook, 'printf "hook\\n" >> "$DSH_ORDER_FILE"\n')
  183. const runnerScript = [
  184. 'const { appendFileSync } = require("node:fs");',
  185. 'const { spawnSync } = require("node:child_process");',
  186. 'appendFileSync(process.env.DSH_ORDER_FILE, "runner\\n");',
  187. 'const child = spawnSync(process.argv[1], process.argv.slice(2), { env: process.env, stdio: "inherit" });',
  188. 'process.exit(child.status ?? 125);',
  189. ].join('')
  190. const { bash } = await setup({}, argv => ({
  191. argv: [process.execPath, '-e', runnerScript, ...argv],
  192. enforcement: 'full',
  193. denialSignatures: UNIX_SIGNATURES,
  194. runnerFailureRules: RUNNER_FAILURE,
  195. }))
  196. try {
  197. const result = await bash.run(bash.resolve({
  198. command: 'true',
  199. env: { BASH_ENV: hook },
  200. dshEnv: { DSH_ORDER_FILE: order },
  201. }))
  202. expect(result.exitCode).toBe(0)
  203. expect(readFileSync(order, 'utf8')).toBe('runner\nhook\n')
  204. } finally {
  205. rmSync(dir, { recursive: true, force: true })
  206. }
  207. })
  208. it('workspace-write rides the policy, workspaceRoot falling back to process.cwd() when not configured', async () => {
  209. const { bash, calls } = await setup({ mode: 'workspace-write' })
  210. const result = await bash.run(bash.resolve({ command: 'true' }))
  211. expect(result.sandbox).toEqual({ mode: 'workspace-write', denied: false, enforcement: 'full' })
  212. expect(calls[0]?.policy).toEqual({ mode: 'workspace-write', workspaceRoot: resolve(process.cwd()) })
  213. })
  214. it('an explicit workspaceRoot on the policy wins', async () => {
  215. const { calls, bash } = await setup({ mode: 'workspace-write', workspaceRoot: '/ws', cwd: tmpdir() })
  216. await bash.run(bash.resolve({ command: 'true' }))
  217. expect(calls[0]?.policy.workspaceRoot).toBe(resolve('/ws'))
  218. })
  219. it('the provider is consulted per wrap (no caching in the consumer): run and start each hand off', async () => {
  220. const { bash, calls } = await setup()
  221. await bash.run(bash.resolve({ command: 'true' }))
  222. const task = await bash.start(bash.resolve({ command: 'true' }))
  223. await task.done
  224. expect(calls).toHaveLength(2)
  225. })
  226. })
  227. describe('fail closed', () => {
  228. it('propagates the provider\'s structured SANDBOX_UNAVAILABLE on run() and start()', async () => {
  229. const { bash } = await setup({}, () => { throw new SandboxUnavailableError('read-only') })
  230. const spec = bash.resolve({ command: 'echo hi' })
  231. await expect(bash.run(spec)).rejects.toMatchObject({ name: 'SandboxUnavailableError', code: SANDBOX_UNAVAILABLE })
  232. await expect(bash.start(spec)).rejects.toThrow(SandboxUnavailableError)
  233. })
  234. it('preserves an already-aborted foreground call as cancellation', async () => {
  235. const { bash } = await setup()
  236. const controller = new AbortController()
  237. const reason = new Error('caller cancelled before spawn')
  238. controller.abort(reason)
  239. await expect(bash.run(bash.resolve({ command: 'true', signal: controller.signal }))).rejects.toBe(reason)
  240. })
  241. it.each(RUNNER_FORMS)(
  242. 'keeps an invalid workdir ordinary with the %s provider-runner form',
  243. async (_form, runner) => {
  244. const { bash } = await setup({}, argv => ({
  245. argv: [runner, ...argv],
  246. enforcement: 'full',
  247. denialSignatures: UNIX_SIGNATURES,
  248. runnerFailureRules: RUNNER_FAILURE,
  249. }))
  250. const parent = mkdtempSync(join(tmpdir(), 'dsh-sandbox-missing-cwd-'))
  251. try {
  252. const failure = await bash.run(bash.resolve({ command: 'true', workdir: join(parent, 'missing') }))
  253. .catch((error: unknown) => error)
  254. expect(failure).toMatchObject({ code: 'ENOENT' })
  255. expect(failure).not.toBeInstanceOf(SandboxUnavailableError)
  256. } finally {
  257. rmSync(parent, { recursive: true, force: true })
  258. }
  259. },
  260. )
  261. it('keeps an invalid workdir ordinary when danger-full-access bypasses the provider', async () => {
  262. const { bash } = await setup({ mode: 'danger-full-access' })
  263. const parent = mkdtempSync(join(tmpdir(), 'dsh-sandbox-missing-cwd-'))
  264. try {
  265. const failure = await bash.run(bash.resolve({ command: 'true', workdir: join(parent, 'missing') }))
  266. .catch((error: unknown) => error)
  267. expect(failure).toMatchObject({ code: 'ENOENT' })
  268. expect(failure).not.toBeInstanceOf(SandboxUnavailableError)
  269. } finally {
  270. rmSync(parent, { recursive: true, force: true })
  271. }
  272. })
  273. it('keeps Node-shaped synchronous ENOEXEC ordinary in run() and start()', async () => {
  274. const runner = join(spillDir, 'malformed-runner')
  275. const { ctx, bash } = await setup({}, argv => ({
  276. argv: [runner, ...argv],
  277. enforcement: 'full',
  278. denialSignatures: UNIX_SIGNATURES,
  279. runnerFailureRules: RUNNER_FAILURE,
  280. }))
  281. vi.spyOn(ctx.subprocess, 'spawn').mockImplementation(() => {
  282. throw Object.assign(new Error('spawn ENOEXEC'), { code: 'ENOEXEC', syscall: 'spawn' })
  283. })
  284. const foreground = await bash.run(bash.resolve({ command: 'true' })).catch((error: unknown) => error)
  285. expect(foreground).toMatchObject({ code: 'ENOEXEC', syscall: 'spawn' })
  286. expect(foreground).not.toBeInstanceOf(SandboxUnavailableError)
  287. let background: unknown
  288. try {
  289. await bash.start(bash.resolve({ command: 'true' }))
  290. } catch (error) {
  291. background = error
  292. }
  293. expect(background).toMatchObject({ code: 'ENOEXEC', syscall: 'spawn' })
  294. expect(background).not.toBeInstanceOf(SandboxUnavailableError)
  295. })
  296. it('classifies a synchronous SubprocessRuntime EACCES with the exact runner path', async () => {
  297. const runner = join(spillDir, 'unexecutable-runner')
  298. const { ctx, bash } = await setup({}, argv => ({
  299. argv: [runner, ...argv],
  300. enforcement: 'full',
  301. denialSignatures: UNIX_SIGNATURES,
  302. runnerFailureRules: RUNNER_FAILURE,
  303. }))
  304. // This pins an alternative SubprocessRuntime's synchronous seam, not the
  305. // shipped local behavior.
  306. vi.spyOn(ctx.subprocess, 'spawn').mockImplementation(() => {
  307. throw Object.assign(new Error('spawn EACCES'), { code: 'EACCES', syscall: 'spawn', path: runner })
  308. })
  309. await expect(bash.run(bash.resolve({ command: 'true' })))
  310. .rejects.toMatchObject({ name: 'SandboxUnavailableError', code: SANDBOX_UNAVAILABLE })
  311. await expect(bash.start(bash.resolve({ command: 'true' })))
  312. .rejects.toThrow(expect.objectContaining({ name: 'SandboxUnavailableError', code: SANDBOX_UNAVAILABLE }))
  313. })
  314. it('keeps a synchronous cwd-owned ENOENT as the original start() error', async () => {
  315. const runner = './sandbox-runner'
  316. const { ctx, bash } = await setup({}, argv => ({
  317. argv: [runner, ...argv],
  318. enforcement: 'full',
  319. denialSignatures: UNIX_SIGNATURES,
  320. runnerFailureRules: RUNNER_FAILURE,
  321. }))
  322. const parent = mkdtempSync(join(tmpdir(), 'dsh-sandbox-missing-cwd-'))
  323. const workdir = join(parent, 'missing')
  324. const failure = Object.assign(new Error('spawn ENOENT'), { code: 'ENOENT', syscall: `spawn ${runner}`, path: runner })
  325. vi.spyOn(ctx.subprocess, 'spawn').mockImplementation(() => { throw failure })
  326. try {
  327. let thrown: unknown
  328. try {
  329. await bash.start(bash.resolve({ command: 'true', workdir }))
  330. } catch (error) {
  331. thrown = error
  332. }
  333. expect(thrown).toBe(failure)
  334. expect(thrown).not.toBeInstanceOf(SandboxUnavailableError)
  335. } finally {
  336. rmSync(parent, { recursive: true, force: true })
  337. }
  338. })
  339. })
  340. describe('danger-full-access', () => {
  341. it('runs unwrapped: the provider is never consulted, facts carry no enforcement', async () => {
  342. const { bash, calls } = await setup({ mode: 'danger-full-access' })
  343. const result = await bash.run(bash.resolve({ command: 'echo free' }))
  344. expect(result.stdout.text).toBe('free\n')
  345. expect(result.sandbox).toEqual({ mode: 'danger-full-access', denied: false })
  346. expect(calls).toHaveLength(0)
  347. })
  348. it('start() passes through unwrapped and stamps nothing at settle', async () => {
  349. const { bash, calls } = await setup({ mode: 'danger-full-access' })
  350. const task = await bash.start(bash.resolve({ command: 'echo free-bg' }))
  351. await task.done
  352. expect(task.sandbox).toBeUndefined()
  353. expect(task.readOutput().delta).toContain('free-bg')
  354. expect(calls).toHaveLength(0)
  355. })
  356. })
  357. describe('per-call sandbox policy (the session and escalation carrier)', () => {
  358. it('exposes the configured default as the capability fact, and resolve() stamps it', async () => {
  359. const { bash } = await setup()
  360. expect(bash.sandboxMode).toBe('read-only')
  361. expect(bash.resolve({ command: 'true' }).sandboxPolicy).toEqual(executionPolicy('read-only'))
  362. })
  363. it('an explicit policy outranks the default at resolve(), and the wrap follows its mode and root', async () => {
  364. const { bash, calls } = await setup()
  365. const explicit = executionPolicy('workspace-write', '/session/project')
  366. expect(bash.resolve({ command: 'true', sandboxPolicy: explicit }).sandboxPolicy).toEqual(explicit)
  367. await bash.run(bash.resolve({ command: 'true', sandboxPolicy: explicit }))
  368. await bash.run(bash.resolve({ command: 'true' }))
  369. expect(calls.map(call => call.policy)).toEqual([explicit, executionPolicy('read-only')])
  370. })
  371. it('an escalated run reports the mode it ACTUALLY ran under', async () => {
  372. const { bash } = await setup()
  373. const result = await bash.run(bash.resolve({ command: 'true', sandboxPolicy: executionPolicy('workspace-write') }))
  374. expect(result.sandbox).toEqual({ mode: 'workspace-write', denied: false, enforcement: 'full' })
  375. })
  376. it('escalating to danger-full-access bypasses the provider entirely — the grant, not a probe, is the authority there', async () => {
  377. const { bash, calls } = await setup()
  378. const result = await bash.run(bash.resolve({ command: 'echo free', sandboxPolicy: executionPolicy('danger-full-access') }))
  379. expect(result.stdout.text).toBe('free\n')
  380. expect(result.sandbox).toEqual({ mode: 'danger-full-access', denied: false })
  381. expect(calls).toHaveLength(0)
  382. })
  383. it('overlapping background jobs settle with their OWN modes (an escalated task next to a default one)', async () => {
  384. // With per-call policy, tasks under different modes are in flight at
  385. // once — anything keyed off the configured default would misreport the
  386. // escalated one at its settle stamp.
  387. const { bash } = await setup()
  388. const escalated = await bash.start(bash.resolve({ command: 'sleep 0.3; echo "x: Permission denied" >&2; exit 1', sandboxPolicy: executionPolicy('workspace-write') }))
  389. const plain = await bash.start(bash.resolve({ command: 'true' }))
  390. await plain.done
  391. await escalated.done
  392. expect(escalated.sandbox).toEqual({ mode: 'workspace-write', denied: true, enforcement: 'full' })
  393. expect(plain.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full' })
  394. })
  395. it('an escalated danger-full-access background job carries no facts (nothing confined it)', async () => {
  396. const { bash, calls } = await setup()
  397. const task = await bash.start(bash.resolve({ command: 'echo bg-free', sandboxPolicy: executionPolicy('danger-full-access') }))
  398. await task.done
  399. expect(task.sandbox).toBeUndefined()
  400. expect(task.readOutput().delta).toContain('bg-free')
  401. expect(calls).toHaveLength(0)
  402. })
  403. })
  404. describe('classifyDenial', () => {
  405. it('never classifies a clean exit or a signal kill as a denial', () => {
  406. expect(classifyDenial(runResult(0, 'Permission denied'), UNIX_SIGNATURES)).toBe(false)
  407. expect(classifyDenial(runResult(null, 'Permission denied'), UNIX_SIGNATURES)).toBe(false)
  408. })
  409. it('classifies failed runs by the wrap\'s own dialect, conservatively', () => {
  410. expect(classifyDenial(runResult(1, 'touch: cannot touch /x: Read-only file system'), UNIX_SIGNATURES)).toBe(true)
  411. expect(classifyDenial(runResult(1, 'sh: /x: Permission denied'), UNIX_SIGNATURES)).toBe(true)
  412. // Bare EPERM is not a Linux runner's dialect: mount/kill/ptrace fail with
  413. // it unsandboxed too, and the mode vocabulary governs file effects only —
  414. // claiming a file denial here would tell the model the sandbox blocked
  415. // something it never governed.
  416. expect(classifyDenial(runResult(1, 'mount: Operation not permitted'), UNIX_SIGNATURES)).toBe(false)
  417. expect(classifyDenial(runResult(1, 'No such file or directory'), UNIX_SIGNATURES)).toBe(false)
  418. })
  419. it('matches exactly the active backend\'s dialect: EPERM classifies under Seatbelt, EACCES does not under bwrap', () => {
  420. // The same stderr flips meaning with the backend: under Seatbelt, EPERM
  421. // text IS how the kernel refuses a governed file write; under bwrap's
  422. // EROFS-only dialect, `Permission denied` is ordinary DAC, not the
  423. // sandbox — per-wrap signatures are what keep both classifications honest.
  424. expect(classifyDenial(runResult(1, 'bash: /etc/x: Operation not permitted'), ['operation not permitted'])).toBe(true)
  425. expect(classifyDenial(runResult(1, 'sh: /x: Permission denied'), ['read-only file system'])).toBe(false)
  426. })
  427. })
  428. describe('result facts', () => {
  429. it.each([126, 127])('keeps a successfully launched wrapped child exit %i as an ordinary outcome', async (exitCode) => {
  430. const { bash } = await setup({}, argv => ({
  431. argv: ['env', ...argv],
  432. enforcement: 'full',
  433. denialSignatures: UNIX_SIGNATURES,
  434. runnerFailureRules: RUNNER_FAILURE,
  435. }))
  436. const result = await bash.run(bash.resolve({ command: `exit ${exitCode}` }))
  437. expect(result.exitCode).toBe(exitCode)
  438. expect(result.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full' })
  439. })
  440. it('reports a real permission failure as a sandbox denial with the mode it ran under', async () => {
  441. const { bash } = await setup()
  442. const deniedRoot = mkdtempSync(join(tmpdir(), 'dsh-sandbox-denied-'))
  443. try {
  444. const lockedDir = join(deniedRoot, 'locked')
  445. mkdirSync(lockedDir)
  446. chmodSync(lockedDir, 0o555)
  447. const result = await bash.run(bash.resolve({ command: `echo x > ${lockedDir}/f` }))
  448. expect(result.exitCode).not.toBe(0)
  449. expect(result.sandbox).toEqual({ mode: 'read-only', denied: true, enforcement: 'full' })
  450. } finally {
  451. rmSync(deniedRoot, { recursive: true, force: true })
  452. }
  453. })
  454. it('carries the provider\'s partial-enforcement fact through unchanged', async () => {
  455. const { bash } = await setup({}, argv => ({ argv: [...argv], enforcement: 'partial', denialSignatures: UNIX_SIGNATURES, runnerFailureRules: RUNNER_FAILURE }))
  456. const result = await bash.run(bash.resolve({ command: 'true' }))
  457. expect(result.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'partial' })
  458. })
  459. })
  460. describe('background sandbox facts', () => {
  461. it.each(RUNNER_FORMS)('keeps an invalid-workdir rejection ordinary for the %s provider-runner form', async (_form, runner) => {
  462. const { bash } = await setup({}, argv => ({
  463. argv: [runner, ...argv],
  464. enforcement: 'full',
  465. denialSignatures: UNIX_SIGNATURES,
  466. runnerFailureRules: RUNNER_FAILURE,
  467. }))
  468. const parent = mkdtempSync(join(tmpdir(), 'dsh-sandbox-missing-cwd-'))
  469. try {
  470. const task = await bash.start(bash.resolve({ command: 'true', workdir: join(parent, 'missing') }))
  471. await task.done
  472. expect(task.status).toBe('killed')
  473. expect(task.readOutput().delta).toContain('subprocess failed before reporting an outcome:')
  474. expect(task.sandbox).toEqual({
  475. mode: 'read-only',
  476. denied: false,
  477. enforcement: 'full',
  478. })
  479. const accounting = (bash as unknown as { processFacts: Map<unknown, unknown> }).processFacts
  480. expect(accounting.size).toBe(0)
  481. } finally {
  482. rmSync(parent, { recursive: true, force: true })
  483. }
  484. })
  485. it('does not invent runner evidence when a provider rejection has no structured reason', async () => {
  486. const { ctx, bash } = await setup()
  487. const emptyReader: SubprocessOutputReader = {
  488. readFrom: () => ({ text: '', nextOffset: 0, lossy: false }),
  489. }
  490. vi.spyOn(ctx.subprocess, 'spawn').mockReturnValue({
  491. control: undefined,
  492. stdin: undefined,
  493. stdout: undefined,
  494. stderr: undefined,
  495. collected: { stdout: emptyReader, stderr: emptyReader },
  496. // Arbitrary subprocess providers can reject without a value or public stage.
  497. // oxlint-disable-next-line typescript/prefer-promise-reject-errors
  498. done: Promise.reject(undefined),
  499. terminate: vi.fn(),
  500. waitForExit: async () => true,
  501. } satisfies SubprocessHandle)
  502. const task = await bash.start(bash.resolve({ command: 'true' }))
  503. await task.done
  504. expect(task.readOutput().delta).toContain('subprocess failed before reporting an outcome: undefined')
  505. expect(task.sandbox).toEqual({
  506. mode: 'read-only',
  507. denied: false,
  508. enforcement: 'full',
  509. })
  510. })
  511. it('stamps a settled denial: nonzero exit + permission stderr under a confined mode', async () => {
  512. const { bash } = await setup()
  513. const task = await bash.start(bash.resolve({ command: 'echo "x: Permission denied" >&2; exit 1' }))
  514. await task.done
  515. expect(task.sandbox).toEqual({ mode: 'read-only', denied: true, enforcement: 'full' })
  516. })
  517. it('a foreground runner failure throws the fail-closed error, never a task result', async () => {
  518. // The wrap's runner prefix on a failed run means the SANDBOX broke and
  519. // the command never ran — the late twin of the confine-time throw, with
  520. // the matched fatal stderr line carried as the cause.
  521. const { bash } = await setup()
  522. const run = bash.run(bash.resolve({ command: 'echo "fake-runner: ruleset rejected" >&2; exit 125' }))
  523. await expect(run).rejects.toThrow(expect.objectContaining({ code: SANDBOX_UNAVAILABLE }))
  524. await expect(run).rejects.toThrow('fake-runner: ruleset rejected')
  525. })
  526. it('a foreground runner failure outranks denial: runner error text may contain denial words', async () => {
  527. const { bash } = await setup()
  528. await expect(bash.run(bash.resolve({ command: 'echo "fake-runner: cannot open rule path: /x: Permission denied" >&2; exit 125' })))
  529. .rejects.toThrow(expect.objectContaining({ code: SANDBOX_UNAVAILABLE }))
  530. })
  531. it('a settled background runner failure stamps runnerFailed (no error channel remains), not denied', async () => {
  532. const { bash } = await setup()
  533. const task = await bash.start(bash.resolve({ command: 'echo "fake-runner: cannot open rule path: /x: Permission denied" >&2; exit 125' }))
  534. await task.done
  535. expect(task.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full', runnerFailed: true })
  536. })
  537. it('overlapping background jobs keep their OWN wrap facts (per-task, not latest-wrap)', async () => {
  538. // Facts belong to each wrap and may vary between calls. The slow task settles after the
  539. // quick task starts; a shared latest-wrap field would classify and stamp it with the wrong
  540. // task's dialect and enforcement.
  541. const wraps: Array<Pick<ConfinedArgv, 'enforcement' | 'denialSignatures'>> = [
  542. { enforcement: 'partial', denialSignatures: ['permission denied'] },
  543. { enforcement: 'full', denialSignatures: ['read-only file system'] },
  544. ]
  545. let call = 0
  546. const { bash } = await setup({}, (argv) => {
  547. const wrap = wraps[Math.min(call++, wraps.length - 1)] as Pick<ConfinedArgv, 'enforcement' | 'denialSignatures'>
  548. return { argv: [...argv], ...wrap, runnerFailureRules: RUNNER_FAILURE }
  549. })
  550. const slow = await bash.start(bash.resolve({ command: 'sleep 0.4; echo "x: Permission denied" >&2; exit 1' }))
  551. const quick = await bash.start(bash.resolve({ command: 'true' }))
  552. await quick.done
  553. await slow.done
  554. expect(slow.sandbox).toEqual({ mode: 'read-only', denied: true, enforcement: 'partial' })
  555. expect(quick.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full' })
  556. })
  557. it('a signal-killed task is never a denial (null exit code)', async () => {
  558. const { bash } = await setup()
  559. const task = await bash.start(bash.resolve({ command: 'echo "Permission denied" >&2; sleep 30' }))
  560. // Let the stderr land before the kill so the classifier sees the
  561. // signature and must still refuse it on the null exit code alone.
  562. await vi.waitFor(() => { expect(task.readOutput().delta).toContain('Permission denied') })
  563. task.kill()
  564. await task.done
  565. expect(task.sandbox).toEqual({ mode: 'read-only', denied: false, enforcement: 'full' })
  566. })
  567. it('disposal kills wrapped background jobs (inherited HMR safety)', async () => {
  568. const { ctx, bash } = await setup()
  569. const task = await bash.start(bash.resolve({ command: 'sleep 30' }))
  570. await ctx.fiber.dispose()
  571. expect(task.status).toBe('killed')
  572. })
  573. })