tools.spec.ts 50 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020102110221023102410251026102710281029103010311032103310341035103610371038103910401041104210431044104510461047104810491050105110521053105410551056105710581059106010611062106310641065106610671068106910701071107210731074107510761077107810791080108110821083108410851086108710881089109010911092109310941095
  1. /**
  2. * Consumer API tests over a fake provider and the real policy collaborator: schemas,
  3. * validation, formatting, typed errors, intent dispatch, and observation-driven authorization.
  4. */
  5. import { describe, expect, it, vi } from 'vitest'
  6. import { Context } from '@deepseek-ai/cordis'
  7. import { PtcRuntime } from '@deepseek-ai/dsh-ptc-runtime'
  8. import { createScope, type Scope } from '@deepseek-ai/dsh-scope'
  9. import { mkdirSync, mkdtempSync, rmSync, symlinkSync } from 'node:fs'
  10. import { tmpdir } from 'node:os'
  11. import { join, sep } from 'node:path'
  12. import { turnBoundaryProjectionDefinition } from '@deepseek-ai/dsh-agent-loop'
  13. import { ToolCallId } from '@deepseek-ai/dsh-llm'
  14. import SystemPrompt, { renderPrompt } from '@deepseek-ai/dsh-system-prompt'
  15. import ToolRuntime, { type ToolResult } from '@deepseek-ai/dsh-tools'
  16. import { FileSystem, FsError, FsTargetKey, FsVersion } from '@deepseek-ai/dsh-fs'
  17. import type {
  18. FsDirEntry,
  19. FsEditOutcome,
  20. FsEditRequest,
  21. FsInfo,
  22. FsPathInfo,
  23. FsTarget,
  24. FsWriteIntent,
  25. FsWriteOutcome,
  26. } from '@deepseek-ai/dsh-fs'
  27. import * as FsPolicy from '@deepseek-ai/dsh-fs-observation-policy'
  28. import * as ToolFs from '@deepseek-ai/dsh-tool-fs'
  29. import { STREAM_MIN_SIZE } from '../src/read.ts'
  30. import { formatReadOutput } from '../src/read-render.ts'
  31. import type { FileReadOutcome } from '../src/read-render.ts'
  32. import { sessionCwd } from '../src/session-cwd.ts'
  33. import ApprovalService from '@deepseek-ai/dsh-user-approval'
  34. import type { SandboxExecutionPolicy, SandboxMode } from '@deepseek-ai/dsh-sandbox'
  35. import SandboxPolicyService from '@deepseek-ai/dsh-sandbox-policy'
  36. import { SessionId, SessionLogOffset, SessionSeq } from '@deepseek-ai/dsh-session'
  37. import SessionProjectionRegistry from '@deepseek-ai/dsh-session-projection'
  38. const testToolSignal = new AbortController().signal
  39. /** An in-memory fake provider; a test can arm a rejection on any primitive. */
  40. class FakeFs extends FileSystem {
  41. files = new Map<string, string>()
  42. rejectWith?: FsError
  43. writeIntents: (FsWriteIntent | undefined)[] = []
  44. editIntents: ({ version: FsVersion } | undefined)[] = []
  45. private throwIfArmed(): void {
  46. if (this.rejectWith) throw this.rejectWith
  47. }
  48. override async resolve(path: string): Promise<FsTarget> {
  49. return { targetKey: FsTargetKey(`key:${path}`), displayPath: `/abs/${path}` }
  50. }
  51. override processPath(target: FsTarget): string { return String(target.targetKey) }
  52. override fileUrl(target: FsTarget): string { return `file://${target.targetKey}` }
  53. override contains(parent: FsTarget, child: FsTarget): boolean {
  54. return child.targetKey === parent.targetKey || String(child.targetKey).startsWith(`${parent.targetKey}/`)
  55. }
  56. override async stat(target: FsTarget): Promise<FsInfo | undefined> {
  57. this.throwIfArmed()
  58. const content = this.files.get(target.targetKey)
  59. if (content === undefined) return undefined
  60. return { version: FsVersion('v1'), type: 'file', size: content.length }
  61. }
  62. override async lstat(path: string): Promise<FsPathInfo | undefined> {
  63. const content = this.files.get(`key:${path}`)
  64. if (content === undefined) return undefined
  65. return { version: FsVersion('v1'), type: 'file', size: content.length }
  66. }
  67. override async readText(target: FsTarget): Promise<string> {
  68. return this.files.get(target.targetKey) ?? ''
  69. }
  70. override async streamText(target: FsTarget): Promise<AsyncIterable<string>> {
  71. const content = this.files.get(target.targetKey) ?? ''
  72. return (async function* () { yield content })()
  73. }
  74. override async readBytes(target: FsTarget, _signal: AbortSignal | undefined, maxBytes: number): Promise<Uint8Array> {
  75. const bytes = new TextEncoder().encode(this.files.get(target.targetKey) ?? '')
  76. if (bytes.length > maxBytes) {
  77. throw new FsError(`too large: ${target.displayPath}`, 'FS_TOO_LARGE')
  78. }
  79. return bytes
  80. }
  81. override async readByteRange(target: FsTarget, range: { offset: number; length: number }): Promise<Uint8Array> {
  82. return new TextEncoder().encode(this.files.get(target.targetKey) ?? '').subarray(range.offset, range.offset + range.length)
  83. }
  84. override async listDir(_target: FsTarget): Promise<FsDirEntry[]> {
  85. return []
  86. }
  87. override async writeText(target: FsTarget, content: string, expected?: FsWriteIntent): Promise<FsWriteOutcome> {
  88. this.throwIfArmed()
  89. this.writeIntents.push(expected)
  90. const before = this.files.get(target.targetKey) ?? null
  91. this.files.set(target.targetKey, content)
  92. return { operation: before !== null ? 'update' : 'create', version: FsVersion('v2'), before, after: content }
  93. }
  94. override async editText(target: FsTarget, edit: FsEditRequest, expected?: { version: FsVersion }): Promise<FsEditOutcome> {
  95. this.throwIfArmed()
  96. this.editIntents.push(expected)
  97. const content = this.files.get(target.targetKey) ?? ''
  98. const after = content.split(edit.oldString).join(edit.newString)
  99. this.files.set(target.targetKey, after)
  100. return { version: FsVersion('v3'), before: content, after }
  101. }
  102. }
  103. async function setup() {
  104. const ctx = new Context()
  105. await ctx.plugin(SystemPrompt)
  106. await ctx.plugin(ToolRuntime)
  107. await ctx.plugin(FakeFs)
  108. await ctx.plugin(FsPolicy)
  109. await ctx.plugin(ToolFs)
  110. const fs = ctx.fs as FakeFs
  111. return { ctx, fs }
  112. }
  113. let callCounter = 0
  114. function call(ctx: Context, name: string, args: unknown, agent?: object) {
  115. return ctx.tools.execute({
  116. signal: testToolSignal,
  117. callId: ToolCallId(`call-${++callCounter}`),
  118. name,
  119. arguments: args,
  120. ...agent ? { agent: agent as never } : {},
  121. })
  122. }
  123. function text(result: { content: { type: string; text?: string }[] }): string {
  124. return result.content.filter(b => b.type === 'text').map(b => b.text).join('')
  125. }
  126. describe('session cwd resolution', () => {
  127. const execution = (cwd?: string) => cwd === undefined
  128. ? {}
  129. : { agent: { session: { header: { cwd } } } }
  130. it('preserves cwd spelling so the filesystem provider resolves parent traversal', () => {
  131. const cwd = process.cwd()
  132. const throughParent = `${cwd}${sep}..`
  133. expect(sessionCwd(execution() as never)).toBeUndefined()
  134. expect(sessionCwd(execution(cwd) as never)).toBe(cwd)
  135. expect(sessionCwd(execution(throughParent) as never)).toBe(throughParent)
  136. const root = mkdtempSync(join(tmpdir(), 'dsh-tool-fs-session-cwd-'))
  137. const physical = join(root, 'physical')
  138. const link = join(root, 'link')
  139. try {
  140. mkdirSync(physical)
  141. symlinkSync(physical, link, process.platform === 'win32' ? 'junction' : 'dir')
  142. expect(sessionCwd(execution(link) as never)).toBe(link)
  143. } finally {
  144. rmSync(root, { recursive: true, force: true })
  145. }
  146. })
  147. })
  148. describe('registration', () => {
  149. it('registers read, write, and edit', async () => {
  150. const { ctx } = await setup()
  151. expect(ctx.tools.schemas().map(s => s.name).sort()).toEqual(['edit', 'read', 'write'])
  152. })
  153. it('declares read parallel-safe while write/edit remain exclusive', async () => {
  154. const { ctx } = await setup()
  155. expect(ctx.tools.executionMode({ signal: testToolSignal, callId: ToolCallId('read-safe'), name: 'read', arguments: { file_path: 'a.txt' } }))
  156. .toEqual({ kind: 'parallel' })
  157. expect(ctx.tools.executionMode({ signal: testToolSignal, callId: ToolCallId('write-exclusive'), name: 'write', arguments: { file_path: 'a.txt', content: 'x' } }))
  158. .toEqual({ kind: 'exclusive' })
  159. expect(ctx.tools.executionMode({ signal: testToolSignal, callId: ToolCallId('edit-exclusive'), name: 'edit', arguments: { file_path: 'a.txt', old_string: 'x', new_string: 'y' } }))
  160. .toEqual({ kind: 'exclusive' })
  161. })
  162. it('registers prompt sections for each tool', async () => {
  163. const { ctx } = await setup()
  164. const prompt = renderPrompt(await ctx.systemPrompt.assemble())
  165. expect(prompt).toContain('Use the read tool')
  166. expect(prompt).toContain('Use the write tool')
  167. expect(prompt).toContain('Use the edit tool')
  168. })
  169. it('stays pending until ctx.fs exists (inject)', async () => {
  170. const ctx = new Context()
  171. await ctx.plugin(SystemPrompt)
  172. await ctx.plugin(ToolRuntime)
  173. await ctx.plugin(ToolFs) // no fs provider
  174. expect(ctx.tools.schemas()).toHaveLength(0)
  175. })
  176. it('unregisters everything on fiber disposal (HMR safety)', async () => {
  177. const ctx = new Context()
  178. await ctx.plugin(SystemPrompt)
  179. await ctx.plugin(ToolRuntime)
  180. await ctx.plugin(FakeFs)
  181. await ctx.plugin(FsPolicy)
  182. const fiber = await ctx.plugin(ToolFs)
  183. // Each tool contributes BOTH a schema and a prompt section; disposal must
  184. // withdraw both, not just the schemas.
  185. expect(ctx.tools.schemas()).toHaveLength(3)
  186. const sectionNames = (a: { sections: { name: string }[] }) => a.sections.map(s => s.name).sort()
  187. expect(sectionNames(await ctx.systemPrompt.assemble())).toEqual(['deployment:persona-prefix', 'deployment:persona-suffix', 'harness:identity', 'tool:edit', 'tool:read', 'tool:write'])
  188. await fiber.dispose()
  189. expect(ctx.tools.schemas()).toHaveLength(0)
  190. // Only the system-prompt plugin's own built-in sections remain.
  191. expect(sectionNames(await ctx.systemPrompt.assemble())).toEqual(['deployment:persona-prefix', 'deployment:persona-suffix', 'harness:identity'])
  192. })
  193. })
  194. describe('read tool', () => {
  195. it('formats line-numbered content with a footer', async () => {
  196. const { ctx, fs } = await setup()
  197. fs.files.set('key:a.txt', 'hello\nworld')
  198. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  199. expect(result.isError).toBe(false)
  200. if (result.isError) throw new Error('expected read success')
  201. expect(result.value).toEqual({
  202. path: '/abs/a.txt',
  203. offset: 1,
  204. lines: [{ number: 1, text: 'hello' }, { number: 2, text: 'world' }],
  205. totalLines: 2,
  206. })
  207. expect(text(result)).toBe(`<path>/abs/a.txt</path>
  208. <type>file</type>
  209. <content>
  210. 1: hello
  211. 2: world
  212. (End of file - total 2 lines)
  213. </content>`)
  214. })
  215. it('returns an explicit empty canonical line window for an empty file', async () => {
  216. const { ctx, fs } = await setup()
  217. fs.files.set('key:empty.txt', '')
  218. const result = await call(ctx, 'read', { file_path: 'empty.txt' })
  219. if (result.isError) throw new Error('expected empty read success')
  220. expect(result.value).toEqual({ path: '/abs/empty.txt', offset: 1, lines: [], totalLines: 0 })
  221. expect(text(result)).toContain('(End of file - total 0 lines)')
  222. })
  223. it('rejects a non-positive offset via arg validation', async () => {
  224. const { ctx } = await setup()
  225. const result = await call(ctx, 'read', { file_path: 'a.txt', offset: 0 })
  226. expect(result.isError).toBe(true)
  227. expect(text(result)).toContain('offset must be a positive integer')
  228. })
  229. it('rejects a fractional offset and a zero/negative limit', async () => {
  230. const { ctx } = await setup()
  231. for (const args of [
  232. { file_path: 'a.txt', offset: 1.5 },
  233. { file_path: 'a.txt', limit: 0 },
  234. { file_path: 'a.txt', limit: -3 },
  235. ]) {
  236. const result = await call(ctx, 'read', args)
  237. expect(result.isError, JSON.stringify(args)).toBe(true)
  238. expect(text(result)).toMatch(/must be a positive integer/)
  239. }
  240. })
  241. it('rejects a non-JSON numeric offset before tool-specific validation', async () => {
  242. const { ctx } = await setup()
  243. const result = await call(ctx, 'read', { file_path: 'a.txt', offset: Number.NaN })
  244. expect(result.isError).toBe(true)
  245. expect(text(result)).toContain('tool execution arguments must be losslessly JSON-serializable')
  246. })
  247. it('rejects a limit above the cap', async () => {
  248. const { ctx } = await setup()
  249. const result = await call(ctx, 'read', { file_path: 'a.txt', limit: 99999 })
  250. expect(result.isError).toBe(true)
  251. expect(text(result)).toContain('less than or equal to 2000')
  252. })
  253. it('rejects a blank file_path', async () => {
  254. const { ctx } = await setup()
  255. const result = await call(ctx, 'read', { file_path: ' ' })
  256. expect(result.isError).toBe(true)
  257. expect(text(result)).toContain('file_path must be a non-empty string')
  258. })
  259. it('records observed state so a follow-up edit by the same session is authorized', async () => {
  260. const { ctx, fs } = await setup()
  261. const session = { header: {} }
  262. fs.files.set('key:a.txt', 'hello')
  263. expect((await call(ctx, 'read', { file_path: 'a.txt' }, { session })).isError).toBe(false)
  264. const edited = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'hello', new_string: 'bye' }, { session })
  265. expect(edited.isError).toBe(false)
  266. expect(fs.editIntents).toEqual([{ version: 'v1' }])
  267. })
  268. it('propagates FS_NOT_FOUND for an absent file', async () => {
  269. const { ctx } = await setup()
  270. const result = await call(ctx, 'read', { file_path: 'missing.txt' })
  271. expect(result.isError).toBe(true)
  272. expect(result.error).toMatchObject({ info: { code: 'FS_NOT_FOUND' } })
  273. })
  274. it('rejects a non-regular target', async () => {
  275. const { ctx, fs } = await setup()
  276. fs.files.set('key:d', '')
  277. fs.stat = async () => ({ version: FsVersion('v1'), type: 'directory' })
  278. const result = await call(ctx, 'read', { file_path: 'd' })
  279. expect(result.isError).toBe(true)
  280. expect(result.error).toMatchObject({ info: { code: 'FS_NOT_REGULAR_FILE' } })
  281. })
  282. it('streams a large file (size at/above the cap) instead of reading whole', async () => {
  283. const { ctx, fs } = await setup()
  284. fs.files.set('key:big.txt', 'alpha\nbeta')
  285. const readSpy = vi.spyOn(fs, 'readText')
  286. const streamSpy = vi.spyOn(fs, 'streamText')
  287. fs.stat = async () => ({ version: FsVersion('v1'), type: 'file', size: STREAM_MIN_SIZE })
  288. const result = await call(ctx, 'read', { file_path: 'big.txt' })
  289. expect(result.isError).toBe(false)
  290. expect(text(result)).toContain('1: alpha')
  291. expect(streamSpy).toHaveBeenCalled()
  292. expect(readSpy).not.toHaveBeenCalled()
  293. })
  294. it('streams when the backend reports no size (never buffers a size-less file)', async () => {
  295. const { ctx, fs } = await setup()
  296. fs.files.set('key:a.txt', 'alpha')
  297. const streamSpy = vi.spyOn(fs, 'streamText')
  298. fs.stat = async () => ({ version: FsVersion('v1'), type: 'file' }) // no size
  299. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  300. expect(result.isError).toBe(false)
  301. expect(streamSpy).toHaveBeenCalled()
  302. })
  303. it('surfaces a byte-capped read as a truncated footer', async () => {
  304. const { ctx, fs } = await setup()
  305. // Many long lines so the window hits the byte cap before EOF.
  306. fs.files.set('key:big.txt', Array.from({ length: 2000 }, () => 'y'.repeat(100)).join('\n'))
  307. const result = await call(ctx, 'read', { file_path: 'big.txt' })
  308. expect(result.isError).toBe(false)
  309. expect(text(result)).toContain('Output capped.')
  310. })
  311. it('attaches the structured window as presentation meta, and presentResult narrows it into a read card', async () => {
  312. const { ctx, fs } = await setup()
  313. fs.files.set('key:a.ts', 'const x = 1\nconst y = 2')
  314. const result = await call(ctx, 'read', { file_path: 'a.ts' })
  315. expect(result.isError).toBe(false)
  316. if (result.isError) throw new Error('expected read success')
  317. // The extension drives the lang hint; the window rides on persisted meta.
  318. expect(result.meta).toEqual({
  319. path: '/abs/a.ts',
  320. offset: 1,
  321. lines: [{ number: 1, text: 'const x = 1' }, { number: 2, text: 'const y = 2' }],
  322. totalLines: 2,
  323. lang: 'ts',
  324. })
  325. const view = ctx.tools.get('read')?.presentResult?.({ file_path: 'a.ts' }, result)
  326. expect(view).toEqual({
  327. card: 'read',
  328. path: '/abs/a.ts',
  329. offset: 1,
  330. lines: [{ number: 1, text: 'const x = 1' }, { number: 2, text: 'const y = 2' }],
  331. totalLines: 2,
  332. lang: 'ts',
  333. content: [{ type: 'text', text: '1: const x = 1\n2: const y = 2\n\n(End of file - total 2 lines)' }],
  334. })
  335. })
  336. it('omits the lang hint in meta for an extension that maps to no language', async () => {
  337. const { ctx, fs } = await setup()
  338. fs.files.set('key:notes', 'plain')
  339. const result = await call(ctx, 'read', { file_path: 'notes' })
  340. if (result.isError) throw new Error('expected read success')
  341. expect(result.meta).toEqual({ path: '/abs/notes', offset: 1, lines: [{ number: 1, text: 'plain' }], totalLines: 1 })
  342. })
  343. })
  344. describe('formatReadOutput footer variants', () => {
  345. const base: FileReadOutcome = { offset: 1, lines: [{ number: 1, text: 'x' }], totalLines: 1 }
  346. it('reports a byte-capped read', () => {
  347. const out = formatReadOutput('/f', { ...base, totalLines: 99, truncatedByBytes: true })
  348. expect(out).toContain('(Output capped. Showing lines 1-1. Use offset=2 to continue.)')
  349. })
  350. it('reports a more-remaining page', () => {
  351. const out = formatReadOutput('/f', { ...base, totalLines: 99 })
  352. expect(out).toContain('(Showing lines 1-1 of 99. Use offset=2 to continue.)')
  353. })
  354. it('reports end-of-file', () => {
  355. expect(formatReadOutput('/f', base)).toContain('(End of file - total 1 lines)')
  356. })
  357. it('renders an empty file as just the footer', () => {
  358. const out = formatReadOutput('/f', { ...base, lines: [], totalLines: 0 })
  359. expect(out).toContain('(End of file - total 0 lines)')
  360. expect(out).not.toContain(': ')
  361. })
  362. })
  363. describe('write tool', () => {
  364. it('formats a create result and uses createIfAbsent (unobserved, with the gate)', async () => {
  365. const { ctx, fs } = await setup()
  366. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'hi' }, { session: { header: {} } })
  367. expect(result.isError).toBe(false)
  368. if (result.isError) throw new Error('expected write success')
  369. expect(result.value).toEqual({ path: '/abs/a.txt', operation: 'create', before: null, after: 'hi' })
  370. expect(text(result)).toContain('Created file')
  371. expect(fs.writeIntents).toEqual([{ kind: 'createIfAbsent' }])
  372. })
  373. it('rejects a blank file_path', async () => {
  374. const { ctx } = await setup()
  375. const result = await call(ctx, 'write', { file_path: ' ', content: 'hi' })
  376. expect(result.isError).toBe(true)
  377. expect(text(result)).toContain('file_path must be a non-empty string')
  378. })
  379. it('propagates a backend FsError as an isError result carrying its code and remedy', async () => {
  380. const { ctx, fs } = await setup()
  381. fs.rejectWith = new FsError('blocked', 'FS_STALE_VERSION')
  382. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'hi' })
  383. expect(result.isError).toBe(true)
  384. expect(result.error).toMatchObject({ info: { name: 'FsError', code: 'FS_STALE_VERSION' } })
  385. expect(text(result)).toContain('re-read the file, then retry')
  386. })
  387. })
  388. describe('edit tool', () => {
  389. it('formats a single-replacement success after a read', async () => {
  390. const { ctx, fs } = await setup()
  391. const session = { header: {} }
  392. fs.files.set('key:a.txt', 'a')
  393. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  394. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'a', new_string: 'b' }, { session })
  395. if (result.isError) throw new Error('expected edit success')
  396. expect(result.value).toEqual({ path: '/abs/a.txt', before: 'a', after: 'b' })
  397. expect(text(result)).toBe('The file /abs/a.txt has been updated successfully.')
  398. })
  399. it('formats the replace_all success message distinctly', async () => {
  400. const { ctx, fs } = await setup()
  401. const session = { header: {} }
  402. fs.files.set('key:a.txt', 'a a a')
  403. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  404. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'a', new_string: 'b', replace_all: true }, { session })
  405. expect(text(result)).toBe('The file /abs/a.txt has been updated. All occurrences were successfully replaced.')
  406. })
  407. it('rejects identical old/new strings', async () => {
  408. const { ctx } = await setup()
  409. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'x', new_string: 'x' })
  410. expect(result.isError).toBe(true)
  411. expect(text(result)).toContain('must differ')
  412. })
  413. it('rejects an empty old_string', async () => {
  414. const { ctx } = await setup()
  415. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: '', new_string: 'x' })
  416. expect(result.isError).toBe(true)
  417. expect(text(result)).toContain('old_string must be a non-empty string')
  418. })
  419. it('rejects a blank file_path', async () => {
  420. const { ctx } = await setup()
  421. const result = await call(ctx, 'edit', { file_path: ' ', old_string: 'a', new_string: 'b' })
  422. expect(result.isError).toBe(true)
  423. expect(text(result)).toContain('file_path must be a non-empty string')
  424. })
  425. it('propagates FS_NOT_OBSERVED when the file was never read (the gate decides)', async () => {
  426. const { ctx, fs } = await setup()
  427. fs.files.set('key:a.txt', 'hello')
  428. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'a', new_string: 'b' }, { session: { header: {} } })
  429. expect(result.isError).toBe(true)
  430. expect(result.error).toMatchObject({ info: { code: 'FS_NOT_OBSERVED' } })
  431. })
  432. })
  433. describe('tool-owned presentation (pure presentCall)', () => {
  434. // presentCall is a pure display function of args (no I/O); it drives the
  435. // card's title/kind and the `locations` a UI follows along to.
  436. const presentCall = async (name: string, args: unknown) => {
  437. const { ctx } = await setup()
  438. return ctx.tools.get(name)?.presentCall?.(args)
  439. }
  440. const presentResult = async (name: string, args: unknown, result: ToolResult) => {
  441. const { ctx } = await setup()
  442. return ctx.tools.get(name)?.presentResult?.(args, result)
  443. }
  444. it('read: generic card titled by file with the read window, read kind, location with the offset line', async () => {
  445. expect(await presentCall('read', { file_path: 'src/a.ts', offset: 12, limit: 40 })).toEqual({
  446. card: 'generic', title: 'Read src/a.ts (12 - 51)', kind: 'read',
  447. locations: [{ path: 'src/a.ts', line: 12 }],
  448. })
  449. })
  450. it('read: bare title and line-1 location when offset/limit are unset', async () => {
  451. expect(await presentCall('read', { file_path: 'a.txt' })).toEqual({
  452. card: 'generic', title: 'Read a.txt', kind: 'read', locations: [{ path: 'a.txt', line: 1 }],
  453. })
  454. })
  455. it('read: completed presentation is a read card carrying the structured window with the envelope stripped', async () => {
  456. // The structured line data rides on persisted meta (the raw output object is
  457. // not on the wire); presentResult narrows it and appends the stripped text as
  458. // the no-capability `content` fallback.
  459. const meta = { path: '/tmp/a.ts', offset: 1, lines: [{ number: 1, text: 'hello' }], totalLines: 1, lang: 'ts' }
  460. expect(await presentResult('read', { file_path: 'a.ts' }, {
  461. content: [{ type: 'text', text: '<path>/tmp/a.ts</path>\n<type>file</type>\n<content>\n1: hello\n\n(End of file - total 1 lines)\n</content>' }],
  462. isError: false,
  463. meta,
  464. })).toEqual({
  465. card: 'read',
  466. path: '/tmp/a.ts',
  467. offset: 1,
  468. lines: [{ number: 1, text: 'hello' }],
  469. totalLines: 1,
  470. lang: 'ts',
  471. content: [{ type: 'text', text: '1: hello\n\n(End of file - total 1 lines)' }],
  472. })
  473. // A window whose extension maps to no language omits `lang` from the card.
  474. expect(await presentResult('read', { file_path: 'notes' }, {
  475. content: [{ type: 'text', text: '<path>/tmp/notes</path>\n<type>file</type>\n<content>\nbody\n</content>' }],
  476. isError: false,
  477. meta: { path: '/tmp/notes', offset: 1, lines: [{ number: 1, text: 'body' }], totalLines: 1 },
  478. })).toEqual({
  479. card: 'read',
  480. path: '/tmp/notes',
  481. offset: 1,
  482. lines: [{ number: 1, text: 'body' }],
  483. totalLines: 1,
  484. content: [{ type: 'text', text: 'body' }],
  485. })
  486. // Malformed envelope text with valid meta still declines (the fallback text is unavailable).
  487. expect(await presentResult('read', { file_path: 'a.ts' }, {
  488. content: [{ type: 'text', text: 'malformed replay' }],
  489. isError: false,
  490. meta,
  491. })).toBeUndefined()
  492. // Valid envelope but absent/malformed meta declines to the generic fallback.
  493. expect(await presentResult('read', { file_path: 'a.ts' }, {
  494. content: [{ type: 'text', text: '<path>/tmp/a.ts</path>\n<type>file</type>\n<content>\n1: hello\n</content>' }],
  495. isError: false,
  496. })).toBeUndefined()
  497. expect(await presentResult('read', { file_path: 'a.ts' }, {
  498. content: [{ type: 'text', text: '<path>/tmp/a.ts</path>\n<type>file</type>\n<content>\n1: hello\n</content>' }],
  499. isError: false,
  500. meta: { path: '/tmp/a.ts', lines: 'nope', totalLines: 1 },
  501. })).toBeUndefined()
  502. })
  503. it('read: completed presentation declines errors and non-single-text content', async () => {
  504. const envelope = '<path>/tmp/a.txt</path>\n<type>file</type>\n<content>\nbody\n</content>'
  505. const meta = { path: '/tmp/a.txt', offset: 1, lines: [{ number: 1, text: 'body' }], totalLines: 1 }
  506. expect(await presentResult('read', { file_path: 'a.txt' }, {
  507. content: [{ type: 'text', text: envelope }],
  508. isError: true,
  509. meta,
  510. })).toBeUndefined()
  511. expect(await presentResult('read', { file_path: 'a.txt' }, {
  512. content: [{ type: 'text', text: envelope }, { type: 'text', text: 'second' }],
  513. isError: false,
  514. meta,
  515. })).toBeUndefined()
  516. expect(await presentResult('read', { file_path: 'a.txt' }, {
  517. content: [{ type: 'reasoning', text: envelope }],
  518. isError: false,
  519. meta,
  520. })).toBeUndefined()
  521. })
  522. it('read: "from line N" window when only offset is set', async () => {
  523. expect(await presentCall('read', { file_path: 'a.txt', offset: 5 })).toEqual({
  524. card: 'generic', title: 'Read a.txt (from line 5)', kind: 'read', locations: [{ path: 'a.txt', line: 5 }],
  525. })
  526. })
  527. it('write: diff card (new-file style, oldText null), location', async () => {
  528. expect(await presentCall('write', { file_path: 'out.txt', content: 'hello' })).toEqual({
  529. card: 'diff', title: 'Write out.txt',
  530. diffs: [{ path: 'out.txt', oldText: null, newText: 'hello' }],
  531. locations: [{ path: 'out.txt' }],
  532. })
  533. })
  534. it('read: a limit with no offset windows from line 1', async () => {
  535. expect(await presentCall('read', { file_path: 'a.txt', limit: 10 })).toEqual({
  536. card: 'generic', title: 'Read a.txt (1 - 10)', kind: 'read', locations: [{ path: 'a.txt', line: 1 }],
  537. })
  538. })
  539. it('edit: an empty old_string maps to oldText null (a whole-file replace diff)', async () => {
  540. // presentCall runs on replay of raw logged args, which parseEditArgs does not
  541. // gate — an empty old_string must still produce a valid diff (oldText null).
  542. expect(await presentCall('edit', { file_path: 'a.txt', old_string: '', new_string: 'seed' })).toEqual({
  543. card: 'diff', title: 'Edit a.txt',
  544. diffs: [{ path: 'a.txt', oldText: null, newText: 'seed' }],
  545. locations: [{ path: 'a.txt' }],
  546. })
  547. })
  548. })
  549. describe('result-time contextual diff (meta + presentResult)', () => {
  550. // An edit records the applied contextual hunk on `tool/result` meta, and the tool's
  551. // presentResult narrows it back into a replayable `diff` result card.
  552. const withContext = 'a\nb\nc\nOLD\nd\ne\nf\n'
  553. it('edit: execute attaches the applied hunk as meta { diffs }', async () => {
  554. const { ctx, fs } = await setup()
  555. const session = { header: {} }
  556. fs.files.set('key:a.txt', withContext)
  557. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  558. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'OLD', new_string: 'NEW' }, { session })
  559. expect(result.isError).toBe(false)
  560. expect(result.meta).toEqual({
  561. diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }],
  562. })
  563. })
  564. it('edit: presentResult turns the meta into a diff result card', async () => {
  565. const { ctx, fs } = await setup()
  566. const session = { header: {} }
  567. fs.files.set('key:a.txt', withContext)
  568. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  569. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'OLD', new_string: 'NEW' }, { session })
  570. const view = ctx.tools.get('edit')?.presentResult?.({ file_path: 'a.txt', old_string: 'OLD', new_string: 'NEW' }, result)
  571. expect(view).toEqual({
  572. card: 'diff', title: 'Edit a.txt',
  573. diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }],
  574. })
  575. })
  576. it('write OVERWRITE: execute attaches a contextual hunk; presentResult renders a diff card', async () => {
  577. const { ctx, fs } = await setup()
  578. const session = { header: {} }
  579. fs.files.set('key:a.txt', withContext)
  580. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  581. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'a\nb\nc\nNEW\nd\ne\nf\n' }, { session })
  582. expect(result.isError).toBe(false)
  583. expect(result.meta).toEqual({ operation: 'update', diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }] })
  584. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'x' }, result)
  585. expect(view).toEqual({ card: 'diff', title: 'Write a.txt', diffs: [{ path: 'a.txt', oldText: 'a\nb\nc\nOLD\nd\ne\nf', newText: 'a\nb\nc\nNEW\nd\ne\nf' }] })
  586. })
  587. it('write CREATE: an empty applied-diff projection still falls back to the whole-file diff card', async () => {
  588. // A create has no prior content, yet the completed replacement view must
  589. // remain a diff instead of clobbering the pending new-file diff with text.
  590. const { ctx } = await setup()
  591. const session = { header: {} }
  592. const result = await call(ctx, 'write', { file_path: 'new.txt', content: 'fresh\n' }, { session })
  593. expect(result.isError).toBe(false)
  594. expect(result.meta).toEqual({ operation: 'create', diffs: [] })
  595. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'new.txt', content: 'fresh\n' }, result)
  596. expect(view).toEqual({ card: 'diff', title: 'Write new.txt', diffs: [{ path: 'new.txt', oldText: null, newText: 'fresh\n' }] })
  597. })
  598. it('write OVERWRITE with identical content: an empty applied-diff projection falls back to a whole-file diff', async () => {
  599. const { ctx, fs } = await setup()
  600. const session = { header: {} }
  601. fs.files.set('key:a.txt', 'same\n')
  602. await call(ctx, 'read', { file_path: 'a.txt' }, { session })
  603. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'same\n' }, { session })
  604. expect(result.isError).toBe(false)
  605. // The operation lets a consumer tell this unchanged overwrite from a create with the same empty hunk list.
  606. expect(result.meta).toEqual({ operation: 'update', diffs: [] })
  607. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'same\n' }, result)
  608. expect(view).toEqual({ card: 'diff', title: 'Write a.txt', diffs: [{ path: 'a.txt', oldText: null, newText: 'same\n' }] })
  609. })
  610. it('presentResult returns undefined on an error result (nothing applied)', async () => {
  611. const { ctx } = await setup()
  612. const errorResult = { content: [{ type: 'text' as const, text: 'Error: boom' }], isError: true }
  613. expect(ctx.tools.get('edit')?.presentResult?.({ file_path: 'a.txt', old_string: 'x', new_string: 'y' }, errorResult)).toBeUndefined()
  614. expect(ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'y' }, errorResult)).toBeUndefined()
  615. })
  616. it('edit presentResult returns undefined on malformed meta (defensive narrowing)', async () => {
  617. // edit has no whole-file fallback (only a literal replacement), so a malformed
  618. // meta yields the generic "updated successfully" rendering.
  619. const { ctx } = await setup()
  620. const badMeta = { content: [{ type: 'text' as const, text: 'ok' }], isError: false, meta: { diffs: 'nope' } }
  621. expect(ctx.tools.get('edit')?.presentResult?.({ file_path: 'a.txt', old_string: 'x', new_string: 'y' }, badMeta)).toBeUndefined()
  622. })
  623. it('write presentResult falls back to a whole-file diff on malformed meta (never leaks the result text)', async () => {
  624. // write always renders a diff card so the completed update can't clobber the
  625. // pending diff with the model-facing text; a malformed meta falls back to the
  626. // args-derived whole-file diff, same as a create.
  627. const { ctx } = await setup()
  628. const badMeta = { content: [{ type: 'text' as const, text: 'ok' }], isError: false, meta: { diffs: 'nope' } }
  629. const view = ctx.tools.get('write')?.presentResult?.({ file_path: 'a.txt', content: 'y' }, badMeta)
  630. expect(view).toEqual({ card: 'diff', title: 'Write a.txt', diffs: [{ path: 'a.txt', oldText: null, newText: 'y' }] })
  631. })
  632. })
  633. describe('read caps are plugin config', () => {
  634. async function setupWith(config: ToolFs.Config) {
  635. const ctx = new Context()
  636. await ctx.plugin(SystemPrompt)
  637. await ctx.plugin(ToolRuntime)
  638. await ctx.plugin(FakeFs)
  639. await ctx.plugin(FsPolicy)
  640. await ctx.plugin(ToolFs, config)
  641. return { ctx, fs: ctx.fs as FakeFs }
  642. }
  643. it('a configured readLimit is both the default and the cap, and the schema names it', async () => {
  644. const { ctx, fs } = await setupWith({ readLimit: 2 })
  645. fs.files.set('key:a.txt', 'one\ntwo\nthree\nfour')
  646. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  647. expect(text(result)).toContain('(Showing lines 1-2 of 4. Use offset=3 to continue.)')
  648. const overCap = await call(ctx, 'read', { file_path: 'a.txt', limit: 3 })
  649. expect(overCap.isError).toBe(true)
  650. expect(text(overCap)).toContain('less than or equal to 2')
  651. const readSchema = ctx.tools.schemas().find(s => s.name === 'read')
  652. expect(JSON.stringify(readSchema)).toContain('Defaults to 2.')
  653. })
  654. it('a configured readMaxLineLength truncates lines at the configured length', async () => {
  655. const { ctx, fs } = await setupWith({ readMaxLineLength: 4 })
  656. fs.files.set('key:a.txt', 'abcdefgh')
  657. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  658. expect(text(result)).toContain('1: abcd... (line truncated to 4 chars)')
  659. })
  660. it('a configured readMaxBytes caps the window at the configured bytes', async () => {
  661. const { ctx, fs } = await setupWith({ readMaxBytes: 9 })
  662. fs.files.set('key:a.txt', 'aaaa\nbbbb\ncccc')
  663. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  664. expect(result.isError).toBe(false)
  665. if (result.isError) throw new Error('expected read success')
  666. expect(result.value).toMatchObject({ totalLines: 3 })
  667. expect(text(result)).toContain('Output capped.')
  668. expect(text(result)).not.toContain('cccc')
  669. })
  670. it('a configured readStreamMinSize routes smaller files to the streaming path', async () => {
  671. const { ctx, fs } = await setupWith({ readStreamMinSize: 5 })
  672. fs.files.set('key:a.txt', 'alpha\nbeta')
  673. const readSpy = vi.spyOn(fs, 'readText')
  674. const streamSpy = vi.spyOn(fs, 'streamText')
  675. const result = await call(ctx, 'read', { file_path: 'a.txt' })
  676. expect(result.isError).toBe(false)
  677. expect(streamSpy).toHaveBeenCalled()
  678. expect(readSpy).not.toHaveBeenCalled()
  679. })
  680. it.each([
  681. ['readLimit', { readLimit: 0 }],
  682. ['readLimit', { readLimit: 2.5 }],
  683. ['readMaxLineLength', { readMaxLineLength: -1 }],
  684. ['readMaxBytes', { readMaxBytes: Number.NaN }],
  685. ['readStreamMinSize', { readStreamMinSize: 0 }],
  686. ] as const)('rejects a non-positive or fractional %s at load', async (name, config) => {
  687. const ctx = new Context()
  688. await ctx.plugin(SystemPrompt)
  689. await ctx.plugin(ToolRuntime)
  690. await ctx.plugin(FakeFs)
  691. await expect(ctx.plugin(ToolFs, config)).rejects.toThrow(new RegExp(`tool-fs: ${name} must be a positive integer`))
  692. })
  693. it('has no default export (namespace plugin export shape)', () => {
  694. expect('default' in ToolFs).toBe(false)
  695. })
  696. })
  697. describe('sandbox escalation API (write/edit)', () => {
  698. /** A confining fake `ctx.fs`: reports a default mode, records each per-call policy, and can arm a sandbox denial. */
  699. class SandboxingFakeFs extends FakeFs {
  700. stamped: (SandboxExecutionPolicy | undefined)[] = []
  701. override get sandboxMode(): SandboxMode {
  702. return 'workspace-write'
  703. }
  704. override async writeText(
  705. target: FsTarget,
  706. content: string,
  707. expected?: FsWriteIntent,
  708. _signal?: AbortSignal,
  709. sandboxPolicy?: SandboxExecutionPolicy,
  710. ): Promise<FsWriteOutcome> {
  711. this.stamped.push(sandboxPolicy)
  712. return super.writeText(target, content, expected)
  713. }
  714. override async editText(
  715. target: FsTarget,
  716. edit: FsEditRequest,
  717. expected?: { version: FsVersion },
  718. _signal?: AbortSignal,
  719. sandboxPolicy?: SandboxExecutionPolicy,
  720. ): Promise<FsEditOutcome> {
  721. this.stamped.push(sandboxPolicy)
  722. return super.editText(target, edit, expected)
  723. }
  724. }
  725. async function setupConfining(opts: { approval?: boolean } = {}) {
  726. const ctx = new Context()
  727. await ctx.plugin(SystemPrompt)
  728. await ctx.plugin(ToolRuntime)
  729. await ctx.plugin(SessionProjectionRegistry)
  730. ctx.sessionProjections.register(turnBoundaryProjectionDefinition)
  731. await ctx.plugin(SandboxPolicyService, { mode: 'workspace-write' })
  732. await ctx.plugin(SandboxingFakeFs)
  733. await ctx.plugin(FsPolicy)
  734. if (opts.approval === true) await ctx.plugin(ApprovalService)
  735. await ctx.plugin(ToolFs)
  736. return { ctx, fs: ctx.fs as SandboxingFakeFs }
  737. }
  738. /** A fake agent whose session records appends (the approval audit trail), mid-turn, carrying the given events for the fold. */
  739. function escalationAgent(records: Array<{ type: string; data?: Record<string, unknown> }> = []): object {
  740. const id = SessionId('sess-fs-esc')
  741. const events: Array<{
  742. type: string
  743. seq: ReturnType<typeof SessionSeq>
  744. time: number
  745. data: Record<string, unknown>
  746. }> = [
  747. { type: 'turn/start', seq: SessionSeq(0), time: 0, data: { turn: 1 } },
  748. ...records.map((record, index) => ({
  749. type: record.type,
  750. seq: SessionSeq(index + 1),
  751. time: index + 1,
  752. data: record.data ?? {},
  753. })),
  754. ]
  755. return {
  756. id,
  757. session: {
  758. id,
  759. header: { version: 0, id, createdAt: 0, cwd: '/session-project', isSeeded: false },
  760. inheritedEventCount: SessionLogOffset(0),
  761. firstLiveSeq: SessionLogOffset(0),
  762. get seq() { return SessionLogOffset(events.length) },
  763. eventAt: (seq: ReturnType<typeof SessionSeq>) => events[seq],
  764. snapshotEvents: (
  765. fromSeq = SessionLogOffset(0),
  766. toSeqExclusive = SessionLogOffset(events.length),
  767. ) => events.slice(fromSeq, toSeqExclusive),
  768. append: (type: string, data: Record<string, unknown>) => {
  769. const event = {
  770. type,
  771. seq: SessionSeq(events.length),
  772. time: events.length,
  773. data,
  774. }
  775. events.push(event)
  776. return event
  777. },
  778. },
  779. }
  780. }
  781. function fsSchema(ctx: Context, name: 'write' | 'edit') {
  782. const schema = ctx.tools.schemas().find(s => s.name === name)
  783. if (!schema) throw new Error(`${name} tool not registered`)
  784. return schema as unknown as { parameters: { properties: Record<string, { enum?: string[] }> } }
  785. }
  786. it('fails load when a confining filesystem has no shared sandbox-policy resolver', async () => {
  787. const ctx = new Context()
  788. await ctx.plugin(SystemPrompt)
  789. await ctx.plugin(ToolRuntime)
  790. await ctx.plugin(SandboxingFakeFs)
  791. await expect(ctx.plugin(ToolFs)).rejects.toThrow('tool-fs: the mounted filesystem confines but ctx.sandboxPolicy is missing')
  792. })
  793. it('advertises no escalation fields under a non-confining backend', async () => {
  794. const { ctx } = await setup()
  795. expect(ctx.fs.sandboxMode).toBeUndefined()
  796. for (const name of ['write', 'edit'] as const) {
  797. const props = fsSchema(ctx, name).parameters.properties
  798. expect(props['sandbox_permissions']).toBeUndefined()
  799. expect(props['justification']).toBeUndefined()
  800. }
  801. })
  802. it('advertises the closed target vocabulary on write and edit under a confining backend', async () => {
  803. const { ctx } = await setupConfining()
  804. for (const name of ['write', 'edit'] as const) {
  805. const props = fsSchema(ctx, name).parameters.properties
  806. expect(props['sandbox_permissions']?.enum).toEqual(['workspace-write', 'danger-full-access'])
  807. expect(props['justification']).toBeDefined()
  808. }
  809. })
  810. it('a plain write stamps the default mode with the calling session root', async () => {
  811. const { ctx, fs } = await setupConfining()
  812. await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent())
  813. expect(fs.stamped).toEqual([{
  814. mode: 'workspace-write',
  815. workspaceRoot: '/session-project',
  816. sessionId: SessionId('sess-fs-esc'),
  817. }])
  818. })
  819. it('a standing session override folds onto the stamp', async () => {
  820. const { ctx, fs } = await setupConfining()
  821. await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent([{ type: 'sandbox/mode', data: { mode: 'read-only' } }]))
  822. expect(fs.stamped).toEqual([{
  823. mode: 'read-only',
  824. workspaceRoot: '/session-project',
  825. sessionId: SessionId('sess-fs-esc'),
  826. }])
  827. })
  828. it('a denied write maps to the shared marker plus the escalation hint (isError)', async () => {
  829. const { ctx, fs } = await setupConfining()
  830. fs.rejectWith = new FsError('denied', 'FS_SANDBOX_DENIED')
  831. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent())
  832. expect(result.isError).toBe(true)
  833. expect(text(result)).toContain('[sandbox: file access denied under workspace-write mode]')
  834. expect(text(result)).toContain('retry this exact operation once with sandbox_permissions')
  835. })
  836. it('a non-FS_SANDBOX_DENIED provider error passes through unchanged', async () => {
  837. const { ctx, fs } = await setupConfining()
  838. fs.rejectWith = new FsError('boom', 'FS_IO_ERROR')
  839. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x' }, escalationAgent())
  840. expect(result.isError).toBe(true)
  841. expect(text(result)).toContain('boom')
  842. expect(text(result)).not.toContain('[sandbox:')
  843. })
  844. it('an approved escalation stamps the granted mode onto that write', async () => {
  845. const { ctx, fs } = await setupConfining({ approval: true })
  846. ctx.on('approval/request', () => Promise.resolve('allowed-once' as const))
  847. // Pass a signal so the escalation ask forwards it to the approval request
  848. // (the request rides the tool-execution abort signal).
  849. await ctx.tools.execute({
  850. callId: ToolCallId('call-fs-esc-grant'),
  851. name: 'write',
  852. arguments: { file_path: 'a.txt', content: 'x', sandbox_permissions: 'danger-full-access', justification: 'the test needs it' },
  853. agent: escalationAgent() as never,
  854. signal: new AbortController().signal,
  855. })
  856. expect(fs.stamped).toEqual([{
  857. mode: 'danger-full-access',
  858. workspaceRoot: '/session-project',
  859. sessionId: SessionId('sess-fs-esc'),
  860. }])
  861. })
  862. it.each(['workspace-write', 'danger-full-access'] as const)('writes under repeated %s without approval', async (mode) => {
  863. const { ctx, fs } = await setupConfining()
  864. const result = await call(ctx, 'write', {
  865. file_path: 'a.txt', content: 'x', sandbox_permissions: mode, justification: 'use the current permissions',
  866. }, escalationAgent([{ type: 'sandbox/mode', data: { mode } }]))
  867. expect(result.isError).toBe(false)
  868. expect(fs.stamped).toEqual([{
  869. mode, workspaceRoot: '/session-project', sessionId: SessionId('sess-fs-esc'),
  870. }])
  871. })
  872. it('a rejected escalation fails closed with its own text and never mutates', async () => {
  873. const { ctx, fs } = await setupConfining({ approval: true })
  874. ctx.on('approval/request', () => Promise.resolve('rejected' as const))
  875. const result = await call(ctx, 'edit', { file_path: 'a.txt', old_string: 'x', new_string: 'y', sandbox_permissions: 'danger-full-access', justification: 'the test needs it' }, escalationAgent())
  876. expect(result.isError).toBe(true)
  877. expect(text(result)).toContain('the user rejected escalating this operation to "danger-full-access"')
  878. expect(fs.stamped).toEqual([])
  879. })
  880. it('escalation without an approval service fails closed', async () => {
  881. const { ctx } = await setupConfining()
  882. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'danger-full-access', justification: 'why' }, escalationAgent())
  883. expect(result.isError).toBe(true)
  884. expect(text(result)).toContain('no approval service is composed')
  885. })
  886. it('escalation with an approval service but no agent fails closed', async () => {
  887. const { ctx } = await setupConfining({ approval: true })
  888. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'danger-full-access', justification: 'why' })
  889. expect(result.isError).toBe(true)
  890. expect(text(result)).toContain('no agent to route it through')
  891. })
  892. it('rejects the escalation argument pairing (one field without the other)', async () => {
  893. const { ctx } = await setupConfining()
  894. const missing = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'workspace-write' }, escalationAgent())
  895. expect(missing.isError).toBe(true)
  896. expect(text(missing)).toContain('sandbox_permissions requires a justification')
  897. })
  898. it('sandbox_permissions under a non-confining backend fails closed (unadvertised field still reaches execute)', async () => {
  899. const { ctx } = await setup()
  900. const result = await call(ctx, 'write', { file_path: 'a.txt', content: 'x', sandbox_permissions: 'workspace-write', justification: 'why' }, escalationAgent())
  901. expect(result.isError).toBe(true)
  902. expect(text(result)).toContain('not available in this composition')
  903. })
  904. })
  905. /** Create a real per-agent scope over the mounted tool plugins. */
  906. async function guidanceScope(ctx: Context) {
  907. const key = {}
  908. let scope!: Scope
  909. await ctx.plugin(Object.assign((inner: Context) => { scope = createScope(inner, key) },
  910. { inject: ['tools', 'systemPrompt'] }))
  911. return { key, scope }
  912. }
  913. const originalGuidance = {
  914. read: 'Use the read tool — not shell commands like cat — to inspect text files. Results include line numbers. Use offset and limit to continue reading large files.',
  915. write: 'Use the write tool to create files or completely replace file contents. Existing files are overwritten, so read an existing file first (the default fs-observation-policy requires it) and prefer edit for targeted changes.',
  916. edit: 'Use the edit tool for targeted changes to existing UTF-8 text files. It replaces literal old_string with new_string; by default old_string must appear exactly once. If old_string appears multiple times, provide a more specific old_string or set replace_all to true. Read the file first (the default fs-observation-policy requires it), unless you just created or edited it in this session.',
  917. }
  918. describe('scope-aware filesystem guidance', () => {
  919. it.each(Array.from({ length: 8 }, (_, mask) => mask))('preserves exact text for visible tools (mask %i)', async (mask) => {
  920. const { ctx } = await setup()
  921. const { key, scope } = await guidanceScope(ctx)
  922. const names = ['read', 'write', 'edit'] as const
  923. const allow = names.filter((_, index) => (mask & (1 << index)) !== 0)
  924. const baseline = withPersona(...names.map(name => originalGuidance[name]))
  925. expect(renderPrompt(await ctx.systemPrompt.assemble())).toBe(baseline)
  926. const release = scope.ctx.tools.restrict({ allow })
  927. try {
  928. const assembly = await ctx.systemPrompt.assemble({ scope: key })
  929. expect(assembly.tools.map(tool => tool.name)).toEqual([...allow].sort())
  930. const expected = withPersona(...allow.map(name => name === 'write' && !allow.includes('edit')
  931. ? originalGuidance.write.replace(' and prefer edit for targeted changes', '')
  932. : originalGuidance[name]))
  933. expect(renderPrompt(assembly)).toBe(expected)
  934. expect(renderPrompt(await ctx.systemPrompt.assemble())).toBe(baseline)
  935. release()
  936. expect(renderPrompt(await ctx.systemPrompt.assemble({ scope: key }))).toBe(baseline)
  937. } finally {
  938. await scope.dispose()
  939. }
  940. })
  941. it('honors deny filters and the existing exemption for own-scope tools', async () => {
  942. const { ctx } = await setup()
  943. const { key, scope } = await guidanceScope(ctx)
  944. const write = ctx.tools.get('write')!
  945. scope.ctx.tools.restrict({ deny: ['write', 'edit'] })
  946. try {
  947. expect(renderPrompt(await ctx.systemPrompt.assemble({ scope: key }))).toBe(withPersona(originalGuidance.read))
  948. const denied = await call(ctx, 'write', { file_path: '/blocked', content: 'blocked' }, key)
  949. expect(denied.isError).toBe(true)
  950. expect(text(denied)).toContain('unknown tool "write"')
  951. scope.ctx.tools.register(write)
  952. const assembly = await ctx.systemPrompt.assemble({ scope: key })
  953. expect(assembly.tools.map(tool => tool.name)).toEqual(['read', 'write'])
  954. expect(renderPrompt(assembly)).toBe(withPersona(originalGuidance.read,
  955. originalGuidance.write.replace(' and prefer edit for targeted changes', '')))
  956. } finally {
  957. await scope.dispose()
  958. }
  959. })
  960. })
  961. /** Preserve the default persona and exact section separators in the oracle. */
  962. function withPersona(...sections: string[]): string {
  963. return ['You are an AI agent powered by DeepSeek Harness.', ...sections].join('\n\n')
  964. }
  965. /** Schema assembly only: these cases never execute user code. */
  966. class GuidancePtcRuntime extends PtcRuntime {
  967. resolve(request: import('@deepseek-ai/dsh-ptc-runtime').PtcRunRequest): import('@deepseek-ai/dsh-ptc-runtime').PtcRunSpec { return { ...request, cwd: request.cwd ?? process.cwd(), timeoutMs: request.timeoutMs ?? 120_000 } }
  968. readonly language = 'typescript'
  969. readonly isolation = 'fake'
  970. run() { return Promise.resolve({ logs: [] }) }
  971. }
  972. describe('scope-aware PTC guidance', () => {
  973. it.each(['ptc', 'both'] as const)('uses capability visibility in %s mode', async (mode) => {
  974. const ctx = new Context()
  975. await ctx.plugin(SystemPrompt)
  976. await ctx.plugin(GuidancePtcRuntime)
  977. await ctx.plugin(ToolRuntime, { mode })
  978. await ctx.plugin(FakeFs)
  979. await ctx.plugin(ToolFs)
  980. const { key, scope } = await guidanceScope(ctx)
  981. try {
  982. const baseline = renderPrompt(await ctx.systemPrompt.assemble({ scope: key }))
  983. const release = scope.ctx.tools.restrict({ allow: ['read'] })
  984. const assembly = await ctx.systemPrompt.assemble({ scope: key })
  985. expect(assembly.tools.map(tool => tool.name)).toEqual(mode === 'ptc' ? ['run_code'] : ['read', 'run_code'])
  986. expect(assembly.sections.filter(section => ['tool:read', 'tool:write', 'tool:edit'].includes(section.name))
  987. .map(section => section.text).filter(Boolean)).toEqual([originalGuidance.read])
  988. expect(renderPrompt(assembly)).toContain(originalGuidance.read)
  989. expect(renderPrompt(assembly)).not.toContain(originalGuidance.write)
  990. expect(renderPrompt(assembly)).not.toContain(originalGuidance.edit)
  991. release()
  992. expect(renderPrompt(await ctx.systemPrompt.assemble({ scope: key }))).toBe(baseline)
  993. } finally {
  994. await scope.dispose()
  995. }
  996. })
  997. })