tools.spec.ts 34 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744
  1. /**
  2. * Consumer-surface tests over a fake provider and the real policy collaborator: schemas,
  3. * validation, formatting, typed errors, intent dispatch, and observation-driven authorization.
  4. */
  5. import { describe, expect, it, vi } from 'vitest'
  6. import { Context } from 'cordis'
  7. import { CallId } from '@deepseek-ai/dsh-llm'
  8. import SystemPrompt, { renderPrompt } from '@deepseek-ai/dsh-system-prompt'
  9. import ToolRegistry from '@deepseek-ai/dsh-tools'
  10. import { FileSystem, FsError, FsTargetKey, FsVersion } from '@deepseek-ai/dsh-fs'
  11. import type {
  12. FsDirEntry,
  13. FsEditOutcome,
  14. FsEditRequest,
  15. FsInfo,
  16. FsPathInfo,
  17. FsTarget,
  18. FsWriteIntent,
  19. FsWriteOutcome,
  20. } from '@deepseek-ai/dsh-fs'
  21. import * as FsPolicy from '@deepseek-ai/dsh-fs-policy'
  22. import * as ToolFs from '@deepseek-ai/dsh-tool-fs'
  23. import { STREAM_MIN_SIZE } from '../src/read.ts'
  24. import { formatReadOutput } from '../src/read-render.ts'
  25. import type { FileReadOutcome } from '../src/read-render.ts'
  26. import ApprovalService from '@deepseek-ai/dsh-user-approval'
  27. import type { SandboxMode } from '@deepseek-ai/dsh-sandbox'
  28. /** An in-memory fake provider; a test can arm a rejection on any primitive. */
  29. class FakeFs extends FileSystem {
  30. files = new Map<string, string>()
  31. rejectWith?: FsError
  32. writeIntents: (FsWriteIntent | undefined)[] = []
  33. editIntents: ({ version: FsVersion } | undefined)[] = []
  34. private throwIfArmed(): void {
  35. if (this.rejectWith) throw this.rejectWith
  36. }
  37. override async resolve(path: string): Promise<FsTarget> {
  38. return { targetKey: FsTargetKey(`key:${path}`), displayPath: `/abs/${path}` }
  39. }
  40. override async stat(target: FsTarget): Promise<FsInfo | undefined> {
  41. this.throwIfArmed()
  42. const content = this.files.get(target.targetKey)
  43. if (content === undefined) return undefined
  44. return { version: FsVersion('v1'), type: 'file', size: content.length }
  45. }
  46. override async lstat(path: string): Promise<FsPathInfo | undefined> {
  47. const content = this.files.get(`key:${path}`)
  48. if (content === undefined) return undefined
  49. return { version: FsVersion('v1'), type: 'file', size: content.length }
  50. }
  51. override async readText(target: FsTarget): Promise<string> {
  52. return this.files.get(target.targetKey) ?? ''
  53. }
  54. override async streamText(target: FsTarget): Promise<AsyncIterable<string>> {
  55. const content = this.files.get(target.targetKey) ?? ''
  56. return (async function* () { yield content })()
  57. }
  58. override async listDir(_target: FsTarget): Promise<FsDirEntry[]> {
  59. return []
  60. }
  61. override async writeText(target: FsTarget, content: string, expected?: FsWriteIntent): Promise<FsWriteOutcome> {
  62. this.throwIfArmed()
  63. this.writeIntents.push(expected)
  64. const before = this.files.get(target.targetKey) ?? null
  65. this.files.set(target.targetKey, content)
  66. return { operation: before !== null ? 'update' : 'create', version: FsVersion('v2'), before, after: content }
  67. }
  68. override async editText(target: FsTarget, edit: FsEditRequest, expected?: { version: FsVersion }): Promise<FsEditOutcome> {
  69. this.throwIfArmed()
  70. this.editIntents.push(expected)
  71. const content = this.files.get(target.targetKey) ?? ''
  72. const after = content.split(edit.oldString).join(edit.newString)
  73. this.files.set(target.targetKey, after)
  74. return { version: FsVersion('v3'), before: content, after }
  75. }
  76. }
  77. async function setup() {
  78. const ctx = new Context()
  79. await ctx.plugin(SystemPrompt)
  80. await ctx.plugin(ToolRegistry)
  81. await ctx.plugin(FakeFs)
  82. await ctx.plugin(FsPolicy)
  83. await ctx.plugin(ToolFs)
  84. const fs = ctx.fs as FakeFs
  85. return { ctx, fs }
  86. }
  87. let callCounter = 0
  88. function call(ctx: Context, name: string, args: unknown, agent?: object) {
  89. return ctx.tools.execute({
  90. callId: CallId(`call-${++callCounter}`),
  91. name,
  92. arguments: args,
  93. ...agent ? { agent: agent as never } : {},
  94. })
  95. }
  96. function text(result: { content: { type: string; text?: string }[] }): string {
  97. return result.content.filter(b => b.type === 'text').map(b => b.text).join('')
  98. }
  99. describe('registration', () => {
  100. it('registers read, write, and edit', async () => {
  101. const { ctx } = await setup()
  102. expect(ctx.tools.schemas().map(s => s.name).sort()).toEqual(['edit', 'read', 'write'])
  103. })
  104. it('declares read parallel-safe while write/edit remain exclusive', async () => {
  105. const { ctx } = await setup()
  106. expect(ctx.tools.executionMode({ callId: CallId('read-safe'), name: 'read', arguments: { file_path: 'a.txt' } }))
  107. .toEqual({ kind: 'parallel' })
  108. expect(ctx.tools.executionMode({ callId: CallId('write-exclusive'), name: 'write', arguments: { file_path: 'a.txt', content: 'x' } }))
  109. .toEqual({ kind: 'exclusive' })
  110. expect(ctx.tools.executionMode({ callId: CallId('edit-exclusive'), name: 'edit', arguments: { file_path: 'a.txt', old_string: 'x', new_string: 'y' } }))
  111. .toEqual({ kind: 'exclusive' })
  112. })
  113. it('registers prompt sections for each tool', async () => {
  114. const { ctx } = await setup()
  115. const prompt = renderPrompt(await ctx.systemPrompt.assemble())
  116. expect(prompt).toContain('Use the read tool')
  117. expect(prompt).toContain('Use the write tool')
  118. expect(prompt).toContain('Use the edit tool')
  119. })
  120. it('stays pending until ctx.fs exists (inject)', async () => {
  121. const ctx = new Context()
  122. await ctx.plugin(SystemPrompt)
  123. await ctx.plugin(ToolRegistry)
  124. await ctx.plugin(ToolFs) // no fs provider
  125. expect(ctx.tools.schemas()).toHaveLength(0)
  126. })
  127. it('unregisters everything on fiber disposal (HMR safety)', async () => {
  128. const ctx = new Context()
  129. await ctx.plugin(SystemPrompt)
  130. await ctx.plugin(ToolRegistry)
  131. await ctx.plugin(FakeFs)
  132. await ctx.plugin(FsPolicy)
  133. const fiber = await ctx.plugin(ToolFs)
  134. // Each tool contributes BOTH a schema and a prompt section; disposal must
  135. // withdraw both, not just the schemas.
  136. expect(ctx.tools.schemas()).toHaveLength(3)
  137. const sectionNames = (a: { sections: { name: string }[] }) => a.sections.map(s => s.name).sort()
  138. expect(sectionNames(await ctx.systemPrompt.assemble())).toEqual(['deployment:persona', 'harness:identity', 'tool:edit', 'tool:read', 'tool:write'])
  139. await fiber.dispose()
  140. expect(ctx.tools.schemas()).toHaveLength(0)
  141. // Only the system-prompt plugin's own built-in sections remain.
  142. expect(sectionNames(await ctx.systemPrompt.assemble())).toEqual(['deployment:persona', 'harness:identity'])
  143. })
  144. })
  145. describe('read tool', () => {
  146. it('formats line-numbered content with a footer', async () => {
  147. const { ctx, fs } = await setup()
  148. fs.files.set('key:a.txt', 'hello\nworld')
  149. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  150. expect(result.isError).toBe(false)
  151. expect(text(result)).toBe(`<path>/abs/a.txt</path>
  152. <type>file</type>
  153. <content>
  154. 1: hello
  155. 2: world
  156. (End of file - total 2 lines)
  157. </content>`)
  158. })
  159. it('rejects a non-positive offset via arg validation', async () => {
  160. const { ctx } = await setup()
  161. const result = await call(ctx, 'read', { file_path: 'a.txt', offset: 0 })
  162. expect(result.isError).toBe(true)
  163. expect(text(result)).toContain('offset must be a positive integer')
  164. })
  165. it('rejects a fractional offset and a zero/negative limit', async () => {
  166. const { ctx } = await setup()
  167. for (const args of [
  168. { file_path: 'a.txt', offset: 1.5 },
  169. { file_path: 'a.txt', limit: 0 },
  170. { file_path: 'a.txt', limit: -3 },
  171. ]) {
  172. const result = await call(ctx, 'read', args)
  173. expect(result.isError, JSON.stringify(args)).toBe(true)
  174. expect(text(result)).toMatch(/must be a positive integer/)
  175. }
  176. })
  177. it('rejects a non-JSON numeric offset before tool-specific validation', async () => {
  178. const { ctx } = await setup()
  179. const result = await call(ctx, 'read', { file_path: 'a.txt', offset: Number.NaN })
  180. expect(result.isError).toBe(true)
  181. expect(text(result)).toContain('tool execution arguments must be losslessly JSON-serializable')
  182. })
  183. it('rejects a limit above the cap', async () => {
  184. const { ctx } = await setup()
  185. const result = await call(ctx, 'read', { file_path: 'a.txt', limit: 99999 })
  186. expect(result.isError).toBe(true)
  187. expect(text(result)).toContain('less than or equal to 2000')
  188. })
  189. it('rejects a blank file_path', async () => {
  190. const { ctx } = await setup()
  191. const result = await call(ctx, 'read', { file_path: ' ' })
  192. expect(result.isError).toBe(true)
  193. expect(text(result)).toContain('file_path must be a non-empty string')
  194. })
  195. it('records observed state so a follow-up edit by the same session is authorized', async () => {
  196. const { ctx, fs } = await setup()
  197. const session = { header: {} }
  198. fs.files.set('key:a.txt', 'hello')
  199. expect((await call(ctx, 'read', { file_path: 'a.txt' }, { session })).isError).toBe(false)
  200. const edited = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'hello', new_string: 'bye' }, { session })
  201. expect(edited.isError).toBe(false)
  202. expect(fs.editIntents).toEqual([{ version: 'v1' }])
  203. })
  204. it('propagates FS_NOT_FOUND for an absent file', async () => {
  205. const { ctx } = await setup()
  206. const result = await call(ctx, 'read', { file_path: 'missing.txt' })
  207. expect(result.isError).toBe(true)
  208. expect(result.error).toMatchObject({ code: 'FS_NOT_FOUND' })
  209. })
  210. it('rejects a non-regular target', async () => {
  211. const { ctx, fs } = await setup()
  212. fs.files.set('key:d', '')
  213. fs.stat = async () => ({ version: FsVersion('v1'), type: 'directory' })
  214. const result = await call(ctx, 'read', { file_path: 'd' })
  215. expect(result.isError).toBe(true)
  216. expect(result.error).toMatchObject({ code: 'FS_NOT_REGULAR_FILE' })
  217. })
  218. it('streams a large file (size at/above the cap) instead of reading whole', async () => {
  219. const { ctx, fs } = await setup()
  220. fs.files.set('key:big.txt', 'alpha\nbeta')
  221. const readSpy = vi.spyOn(fs, 'readText')
  222. const streamSpy = vi.spyOn(fs, 'streamText')
  223. fs.stat = async () => ({ version: FsVersion('v1'), type: 'file', size: STREAM_MIN_SIZE })
  224. const result = await call(ctx, 'read', { file_path: 'big.txt' })
  225. expect(result.isError).toBe(false)
  226. expect(text(result)).toContain('1: alpha')
  227. expect(streamSpy).toHaveBeenCalled()
  228. expect(readSpy).not.toHaveBeenCalled()
  229. })
  230. it('streams when the backend reports no size (never buffers a size-less file)', async () => {
  231. const { ctx, fs } = await setup()
  232. fs.files.set('key:a.txt', 'alpha')
  233. const streamSpy = vi.spyOn(fs, 'streamText')
  234. fs.stat = async () => ({ version: FsVersion('v1'), type: 'file' }) // no size
  235. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  236. expect(result.isError).toBe(false)
  237. expect(streamSpy).toHaveBeenCalled()
  238. })
  239. it('surfaces a byte-capped read as a truncated footer', async () => {
  240. const { ctx, fs } = await setup()
  241. // Many long lines so the window hits the byte cap before EOF.
  242. fs.files.set('key:big.txt', Array.from({ length: 2000 }, () => 'y'.repeat(100)).join('\n'))
  243. const result = await call(ctx, 'read', { file_path: 'big.txt' })
  244. expect(result.isError).toBe(false)
  245. expect(text(result)).toContain('Output capped.')
  246. })
  247. })
  248. describe('formatReadOutput footer variants', () => {
  249. const base: FileReadOutcome = { offset: 1, lines: [{ number: 1, text: 'x' }], totalLines: 1 }
  250. it('reports a byte-capped read', () => {
  251. const out = formatReadOutput('/f', { ...base, totalLines: 99, truncatedByBytes: true })
  252. expect(out).toContain('(Output capped. Showing lines 1-1. Use offset=2 to continue.)')
  253. })
  254. it('reports a more-remaining page', () => {
  255. const out = formatReadOutput('/f', { ...base, totalLines: 99 })
  256. expect(out).toContain('(Showing lines 1-1 of 99. Use offset=2 to continue.)')
  257. })
  258. it('reports end-of-file', () => {
  259. expect(formatReadOutput('/f', base)).toContain('(End of file - total 1 lines)')
  260. })
  261. it('renders an empty file as just the footer', () => {
  262. const out = formatReadOutput('/f', { ...base, lines: [], totalLines: 0 })
  263. expect(out).toContain('(End of file - total 0 lines)')
  264. expect(out).not.toContain(': ')
  265. })
  266. })
  267. describe('write tool', () => {
  268. it('formats a create result and uses createIfAbsent (unobserved, with the gate)', async () => {
  269. const { ctx, fs } = await setup()
  270. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'hi' }, { session: { header: {} } })
  271. expect(result.isError).toBe(false)
  272. expect(text(result)).toContain('Created file')
  273. expect(fs.writeIntents).toEqual([{ kind: 'createIfAbsent' }])
  274. })
  275. it('rejects a blank file_path', async () => {
  276. const { ctx } = await setup()
  277. const result = await call(ctx, 'write', { file_path: ' ', content: 'hi' })
  278. expect(result.isError).toBe(true)
  279. expect(text(result)).toContain('file_path must be a non-empty string')
  280. })
  281. it('propagates a backend FsError as an isError result carrying its code', async () => {
  282. const { ctx, fs } = await setup()
  283. fs.rejectWith = new FsError('blocked', 'FS_STALE_VERSION')
  284. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'hi' })
  285. expect(result.isError).toBe(true)
  286. expect(result.error).toMatchObject({ name: 'FsError', code: 'FS_STALE_VERSION' })
  287. })
  288. })
  289. describe('edit tool', () => {
  290. it('formats a single-replacement success after a read', async () => {
  291. const { ctx, fs } = await setup()
  292. const session = { header: {} }
  293. fs.files.set('key:a.txt', 'a')
  294. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  295. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'a', new_string: 'b' }, { session })
  296. expect(text(result)).toBe('The file /abs/a.txt has been updated successfully.')
  297. })
  298. it('formats the replace_all success message distinctly', async () => {
  299. const { ctx, fs } = await setup()
  300. const session = { header: {} }
  301. fs.files.set('key:a.txt', 'a a a')
  302. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  303. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'a', new_string: 'b', replace_all: true }, { session })
  304. expect(text(result)).toBe('The file /abs/a.txt has been updated. All occurrences were successfully replaced.')
  305. })
  306. it('rejects identical old/new strings', async () => {
  307. const { ctx } = await setup()
  308. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'x', new_string: 'x' })
  309. expect(result.isError).toBe(true)
  310. expect(text(result)).toContain('must differ')
  311. })
  312. it('rejects an empty old_string', async () => {
  313. const { ctx } = await setup()
  314. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: '', new_string: 'x' })
  315. expect(result.isError).toBe(true)
  316. expect(text(result)).toContain('old_string must be a non-empty string')
  317. })
  318. it('rejects a blank file_path', async () => {
  319. const { ctx } = await setup()
  320. const result = await call(ctx, 'edit', { file_path: ' ', old_string: 'a', new_string: 'b' })
  321. expect(result.isError).toBe(true)
  322. expect(text(result)).toContain('file_path must be a non-empty string')
  323. })
  324. it('propagates FS_NOT_OBSERVED when the file was never read (the gate decides)', async () => {
  325. const { ctx, fs } = await setup()
  326. fs.files.set('key:a.txt', 'hello')
  327. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'a', new_string: 'b' }, { session: { header: {} } })
  328. expect(result.isError).toBe(true)
  329. expect(result.error).toMatchObject({ code: 'FS_NOT_OBSERVED' })
  330. })
  331. })
  332. describe('tool-owned presentation (pure presentCall)', () => {
  333. // presentCall is a pure display function of args (no I/O); it drives the ACP
  334. // card's title/kind and the `locations` an editor follows along to.
  335. const presentCall = async (name: string, args: unknown) => {
  336. const { ctx } = await setup()
  337. return ctx.tools.get(name)?.presentCall?.(args)
  338. }
  339. it('read: generic card titled by file with the read window, read kind, location with the offset line', async () => {
  340. expect(await presentCall('read', { file_path: 'src/a.ts', offset: 12, limit: 40 })).toEqual({
  341. card: 'generic', title: 'Read src/a.ts (12 - 51)', kind: 'read',
  342. locations: [{ path: 'src/a.ts', line: 12 }],
  343. })
  344. })
  345. it('read: bare title and line-1 location when offset/limit are unset', async () => {
  346. expect(await presentCall('read', { file_path: 'a.txt' })).toEqual({
  347. card: 'generic', title: 'Read a.txt', kind: 'read', locations: [{ path: 'a.txt', line: 1 }],
  348. })
  349. })
  350. it('read: "from line N" window when only offset is set', async () => {
  351. expect(await presentCall('read', { file_path: 'a.txt', offset: 5 })).toEqual({
  352. card: 'generic', title: 'Read a.txt (from line 5)', kind: 'read', locations: [{ path: 'a.txt', line: 5 }],
  353. })
  354. })
  355. it('write: diff card (new-file style, oldText null), location', async () => {
  356. expect(await presentCall('write', { file_path: 'out.txt', content: 'hello' })).toEqual({
  357. card: 'diff', title: 'Write out.txt',
  358. diffs: [{ path: 'out.txt', oldText: null, newText: 'hello' }],
  359. locations: [{ path: 'out.txt' }],
  360. })
  361. })
  362. it('read: a limit with no offset windows from line 1', async () => {
  363. expect(await presentCall('read', { file_path: 'a.txt', limit: 10 })).toEqual({
  364. card: 'generic', title: 'Read a.txt (1 - 10)', kind: 'read', locations: [{ path: 'a.txt', line: 1 }],
  365. })
  366. })
  367. it('edit: an empty old_string maps to oldText null (a whole-file replace diff)', async () => {
  368. // presentCall runs on replay of raw logged args, which parseEditArgs does not
  369. // gate — an empty old_string must still produce a valid diff (oldText null).
  370. expect(await presentCall('edit', { file_path: 'a.txt', old_string: '', new_string: 'seed' })).toEqual({
  371. card: 'diff', title: 'Edit a.txt',
  372. diffs: [{ path: 'a.txt', oldText: null, newText: 'seed' }],
  373. locations: [{ path: 'a.txt' }],
  374. })
  375. })
  376. })
  377. describe('result-time contextual diff (meta + presentResult)', () => {
  378. // An edit records the applied contextual hunk on `tool/result` meta, and the tool's
  379. // presentResult narrows it back into a `diff` result card the bridge renders.
  380. const withContext = 'a\nb\nc\nOLD\nd\ne\nf\n'
  381. it('edit: execute attaches the applied hunk as meta { diffs }', async () => {
  382. const { ctx, fs } = await setup()
  383. const session = { header: {} }
  384. fs.files.set('key:a.txt', withContext)
  385. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  386. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'OLD', new_string: 'NEW' }, { session })
  387. expect(result.isError).toBe(false)
  388. expect(result.meta).toEqual({
  389. diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }],
  390. })
  391. })
  392. it('edit: presentResult turns the meta into a diff result card', async () => {
  393. const { ctx, fs } = await setup()
  394. const session = { header: {} }
  395. fs.files.set('key:a.txt', withContext)
  396. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  397. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'OLD', new_string: 'NEW' }, { session })
  398. const view = ctx.tools.get('edit')?.presentResult?.({ file_path: 'a.txt', old_string: 'OLD', new_string: 'NEW' }, result)
  399. expect(view).toEqual({
  400. card: 'diff', title: 'Edit a.txt',
  401. diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }],
  402. })
  403. })
  404. it('write OVERWRITE: execute attaches a contextual hunk; presentResult renders a diff card', async () => {
  405. const { ctx, fs } = await setup()
  406. const session = { header: {} }
  407. fs.files.set('key:a.txt', withContext)
  408. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  409. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'a\nb\nc\nNEW\nd\ne\nf\n' }, { session })
  410. expect(result.isError).toBe(false)
  411. expect(result.meta).toEqual({ diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }] })
  412. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'x' }, result)
  413. expect(view).toEqual({ card: 'diff', title: 'Write a.txt', diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }] })
  414. })
  415. it('write CREATE: no before-version → no meta, but presentResult still renders a whole-file diff card', async () => {
  416. // A create has no prior content (no `meta`), yet the completed card must be a `diff` — an
  417. // ACP tool_call_update.content REPLACES the call's content, so a non-diff result would
  418. // clobber the pending new-file diff.
  419. const { ctx } = await setup()
  420. const session = { header: {} }
  421. const result = await call(ctx, 'write', { file_path: 'new.txt', content: 'fresh\n' }, { session })
  422. expect(result.isError).toBe(false)
  423. expect(result.meta).toBeUndefined()
  424. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'new.txt', content: 'fresh\n' }, result)
  425. expect(view).toEqual({ card: 'diff', title: 'Write new.txt', diffs: [{ path: 'new.txt', oldText: null, newText: 'fresh\n' }] })
  426. })
  427. it('write OVERWRITE with identical content: a before exists but yields no hunk → no meta, presentResult falls back to a whole-file diff', async () => {
  428. const { ctx, fs } = await setup()
  429. const session = { header: {} }
  430. fs.files.set('key:a.txt', 'same\n')
  431. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  432. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'same\n' }, { session })
  433. expect(result.isError).toBe(false)
  434. expect(result.meta).toBeUndefined()
  435. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'same\n' }, result)
  436. expect(view).toEqual({ card: 'diff', title: 'Write a.txt', diffs: [{ path: 'a.txt', oldText: null, newText: 'same\n' }] })
  437. })
  438. it('presentResult returns undefined on an error result (nothing applied)', async () => {
  439. const { ctx } = await setup()
  440. const errorResult = { content: [{ type: 'text' as const, text: 'Error: boom' }], isError: true }
  441. expect(ctx.tools.get('edit')?.presentResult?.({ file_path: 'a.txt', old_string: 'x', new_string: 'y' }, errorResult)).toBeUndefined()
  442. expect(ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'y' }, errorResult)).toBeUndefined()
  443. })
  444. it('edit presentResult returns undefined on malformed meta (defensive narrowing)', async () => {
  445. // edit has no whole-file fallback (only a literal replacement), so a malformed
  446. // meta yields the generic "updated successfully" rendering.
  447. const { ctx } = await setup()
  448. const badMeta = { content: [{ type: 'text' as const, text: 'ok' }], isError: false, meta: { diffs: 'nope' } }
  449. expect(ctx.tools.get('edit')?.presentResult?.({ file_path: 'a.txt', old_string: 'x', new_string: 'y' }, badMeta)).toBeUndefined()
  450. })
  451. it('write presentResult falls back to a whole-file diff on malformed meta (never leaks the result text)', async () => {
  452. // write always renders a diff card so the completed update can't clobber the
  453. // pending diff with the model-facing text; a malformed meta falls back to the
  454. // args-derived whole-file diff, same as a create.
  455. const { ctx } = await setup()
  456. const badMeta = { content: [{ type: 'text' as const, text: 'ok' }], isError: false, meta: { diffs: 'nope' } }
  457. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'y' }, badMeta)
  458. expect(view).toEqual({ card: 'diff', title: 'Write a.txt', diffs: [{ path: 'a.txt', oldText: null, newText: 'y' }] })
  459. })
  460. })
  461. describe('read caps are plugin config', () => {
  462. async function setupWith(config: ToolFs.Config) {
  463. const ctx = new Context()
  464. await ctx.plugin(SystemPrompt)
  465. await ctx.plugin(ToolRegistry)
  466. await ctx.plugin(FakeFs)
  467. await ctx.plugin(FsPolicy)
  468. await ctx.plugin(ToolFs, config)
  469. return { ctx, fs: ctx.fs as FakeFs }
  470. }
  471. it('a configured readLimit is both the default and the cap, and the schema names it', async () => {
  472. const { ctx, fs } = await setupWith({ readLimit: 2 })
  473. fs.files.set('key:a.txt', 'one\ntwo\nthree\nfour')
  474. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  475. expect(text(result)).toContain('(Showing lines 1-2 of 4. Use offset=3 to continue.)')
  476. const overCap = await call(ctx, 'read', { file_path: 'a.txt', limit: 3 })
  477. expect(overCap.isError).toBe(true)
  478. expect(text(overCap)).toContain('less than or equal to 2')
  479. const readSchema = ctx.tools.schemas().find(s => s.name === 'read')
  480. expect(JSON.stringify(readSchema)).toContain('Defaults to 2.')
  481. })
  482. it('a configured readMaxLineLength truncates lines at the configured length', async () => {
  483. const { ctx, fs } = await setupWith({ readMaxLineLength: 4 })
  484. fs.files.set('key:a.txt', 'abcdefgh')
  485. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  486. expect(text(result)).toContain('1: abcd... (line truncated to 4 chars)')
  487. })
  488. it('a configured readMaxBytes caps the window at the configured bytes', async () => {
  489. const { ctx, fs } = await setupWith({ readMaxBytes: 9 })
  490. fs.files.set('key:a.txt', 'aaaa\nbbbb\ncccc')
  491. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  492. expect(text(result)).toContain('Output capped.')
  493. expect(text(result)).not.toContain('cccc')
  494. })
  495. it('a configured readStreamMinSize routes smaller files to the streaming path', async () => {
  496. const { ctx, fs } = await setupWith({ readStreamMinSize: 5 })
  497. fs.files.set('key:a.txt', 'alpha\nbeta')
  498. const readSpy = vi.spyOn(fs, 'readText')
  499. const streamSpy = vi.spyOn(fs, 'streamText')
  500. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  501. expect(result.isError).toBe(false)
  502. expect(streamSpy).toHaveBeenCalled()
  503. expect(readSpy).not.toHaveBeenCalled()
  504. })
  505. it.each([
  506. ['readLimit', { readLimit: 0 }],
  507. ['readLimit', { readLimit: 2.5 }],
  508. ['readMaxLineLength', { readMaxLineLength: -1 }],
  509. ['readMaxBytes', { readMaxBytes: Number.NaN }],
  510. ['readStreamMinSize', { readStreamMinSize: 0 }],
  511. ] as const)('rejects a non-positive or fractional %s at load', async (name, config) => {
  512. const ctx = new Context()
  513. await ctx.plugin(SystemPrompt)
  514. await ctx.plugin(ToolRegistry)
  515. await ctx.plugin(FakeFs)
  516. await expect(ctx.plugin(ToolFs, config)).rejects.toThrow(new RegExp(`tool-fs: ${name} must be a positive integer`))
  517. })
  518. it('has no default export (namespace plugin export shape)', () => {
  519. expect('default' in ToolFs).toBe(false)
  520. })
  521. })
  522. describe('sandbox escalation surface (write/edit)', () => {
  523. /** A confining fake `ctx.fs`: reports a default mode, records the per-call mode stamped, and can arm a sandbox denial. */
  524. class SandboxingFakeFs extends FakeFs {
  525. stamped: (SandboxMode | undefined)[] = []
  526. override get sandboxMode(): SandboxMode {
  527. return 'workspace-write'
  528. }
  529. override async writeText(
  530. target: FsTarget,
  531. content: string,
  532. expected?: FsWriteIntent,
  533. _signal?: AbortSignal,
  534. sandboxMode?: SandboxMode,
  535. ): Promise<FsWriteOutcome> {
  536. this.stamped.push(sandboxMode)
  537. return super.writeText(target, content, expected)
  538. }
  539. override async editText(
  540. target: FsTarget,
  541. edit: FsEditRequest,
  542. expected?: { version: FsVersion },
  543. _signal?: AbortSignal,
  544. sandboxMode?: SandboxMode,
  545. ): Promise<FsEditOutcome> {
  546. this.stamped.push(sandboxMode)
  547. return super.editText(target, edit, expected)
  548. }
  549. }
  550. async function setupConfining(opts: { approval?: boolean } = {}) {
  551. const ctx = new Context()
  552. await ctx.plugin(SystemPrompt)
  553. await ctx.plugin(ToolRegistry)
  554. await ctx.plugin(SandboxingFakeFs)
  555. await ctx.plugin(FsPolicy)
  556. if (opts.approval === true) await ctx.plugin(ApprovalService)
  557. await ctx.plugin(ToolFs)
  558. return { ctx, fs: ctx.fs as SandboxingFakeFs }
  559. }
  560. /** A fake agent whose session records appends (the approval audit surface), mid-turn, carrying the given events for the fold. */
  561. function escalationAgent(events: Array<{ type: string; data?: Record<string, unknown> }> = []): object {
  562. return {
  563. id: 'agent-fs-esc',
  564. session: {
  565. header: { version: 0, id: 'sess-fs-esc', createdAt: 0 },
  566. events: [{ type: 'turn/start' }, ...events],
  567. append: (type: string, data: Record<string, unknown>) => { events.push({ type, data }) },
  568. },
  569. }
  570. }
  571. function fsSchema(ctx: Context, name: 'write' | 'edit') {
  572. const schema = ctx.tools.schemas().find(s => s.name === name)
  573. if (!schema) throw new Error(`${name} tool not registered`)
  574. return schema as unknown as { parameters: { properties: Record<string, { enum?: string[] }> } }
  575. }
  576. it('advertises no escalation fields under a non-confining backend', async () => {
  577. const { ctx } = await setup()
  578. expect(ctx.fs.sandboxMode).toBeUndefined()
  579. for (const name of ['write', 'edit'] as const) {
  580. const props = fsSchema(ctx, name).parameters.properties
  581. expect(props['sandbox_permissions']).toBeUndefined()
  582. expect(props['justification']).toBeUndefined()
  583. }
  584. })
  585. it('advertises the closed target vocabulary on write and edit under a confining backend', async () => {
  586. const { ctx } = await setupConfining()
  587. for (const name of ['write', 'edit'] as const) {
  588. const props = fsSchema(ctx, name).parameters.properties
  589. expect(props['sandbox_permissions']?.enum).toEqual(['workspace-write', 'danger-full-access'])
  590. expect(props['justification']).toBeDefined()
  591. }
  592. })
  593. it('a plain write stamps nothing (backend default) and no session override folds without one', async () => {
  594. const { ctx, fs } = await setupConfining()
  595. await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent())
  596. expect(fs.stamped).toEqual([undefined])
  597. })
  598. it('a standing session override folds onto the stamp', async () => {
  599. const { ctx, fs } = await setupConfining()
  600. await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent([{ type: 'sandbox/mode', data: { mode: 'read-only' } }]))
  601. expect(fs.stamped).toEqual(['read-only'])
  602. })
  603. it('a denied write maps to the shared marker plus the escalation hint (isError)', async () => {
  604. const { ctx, fs } = await setupConfining()
  605. fs.rejectWith = new FsError('denied', 'FS_SANDBOX_DENIED')
  606. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent())
  607. expect(result.isError).toBe(true)
  608. expect(text(result)).toContain('[sandbox: file access denied under workspace-write mode]')
  609. expect(text(result)).toContain('retry this exact operation once with sandbox_permissions')
  610. })
  611. it('a non-FS_SANDBOX_DENIED provider error passes through unchanged', async () => {
  612. const { ctx, fs } = await setupConfining()
  613. fs.rejectWith = new FsError('boom', 'FS_IO_ERROR')
  614. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent())
  615. expect(result.isError).toBe(true)
  616. expect(text(result)).toContain('boom')
  617. expect(text(result)).not.toContain('[sandbox:')
  618. })
  619. it('an approved escalation stamps the granted mode onto that write', async () => {
  620. const { ctx, fs } = await setupConfining({ approval: true })
  621. ctx.on('approval/request', () => Promise.resolve('allowed-once' as const))
  622. // Pass a signal so the escalation ask forwards it to the approval request
  623. // (the request rides the tool-execution abort signal).
  624. await ctx.tools.execute({
  625. callId: CallId('call-fs-esc-grant'),
  626. name: 'write',
  627. arguments: { file_path: 'a.txt', content: 'x', sandbox_permissions: 'danger-full-access', justification: 'the test needs it' },
  628. agent: escalationAgent() as never,
  629. signal: new AbortController().signal,
  630. })
  631. expect(fs.stamped).toEqual(['danger-full-access'])
  632. })
  633. it('a rejected escalation fails closed with its own text and never mutates', async () => {
  634. const { ctx, fs } = await setupConfining({ approval: true })
  635. ctx.on('approval/request', () => Promise.resolve('rejected' as const))
  636. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'x', new_string: 'y', sandbox_permissions: 'danger-full-access', justification: 'the test needs it' }, escalationAgent())
  637. expect(result.isError).toBe(true)
  638. expect(text(result)).toContain('the user rejected escalating this operation to "danger-full-access"')
  639. expect(fs.stamped).toEqual([])
  640. })
  641. it('escalation without an approval service fails closed', async () => {
  642. const { ctx } = await setupConfining()
  643. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'danger-full-access', justification: 'why' }, escalationAgent())
  644. expect(result.isError).toBe(true)
  645. expect(text(result)).toContain('no approval service is composed')
  646. })
  647. it('escalation with an approval service but no agent fails closed', async () => {
  648. const { ctx } = await setupConfining({ approval: true })
  649. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'danger-full-access', justification: 'why' })
  650. expect(result.isError).toBe(true)
  651. expect(text(result)).toContain('no agent to route it through')
  652. })
  653. it('rejects the escalation argument pairing (one field without the other)', async () => {
  654. const { ctx } = await setupConfining()
  655. const missing = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'workspace-write' }, escalationAgent())
  656. expect(missing.isError).toBe(true)
  657. expect(text(missing)).toContain('sandbox_permissions requires a justification')
  658. })
  659. it('sandbox_permissions under a non-confining backend fails closed (unadvertised field still reaches execute)', async () => {
  660. const { ctx } = await setup()
  661. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'workspace-write', justification: 'why' }, escalationAgent())
  662. expect(result.isError).toBe(true)
  663. expect(text(result)).toContain('not available in this composition')
  664. })
  665. })