code-mode.spec.ts 85 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061106210631064106510661067106810691070107110721073107410751076107710781079108010811082108310841085108610871088108910901091109210931094109510961097109810991100110111021103110411051106110711081109111011111112111311141115111611171118111911201121112211231124112511261127112811291130113111321133113411351136113711381139114011411142114311441145114611471148114911501151115211531154115511561157115811591160116111621163116411651166116711681169117011711172117311741175117611771178117911801181118211831184118511861187118811891190119111921193119411951196119711981199120012011202120312041205120612071208120912101211121212131214121512161217121812191220122112221223122412251226122712281229123012311232123312341235123612371238123912401241124212431244124512461247124812491250125112521253125412551256125712581259126012611262126312641265126612671268126912701271127212731274127512761277127812791280128112821283128412851286128712881289129012911292129312941295129612971298129913001301130213031304130513061307130813091310131113121313131413151316131713181319132013211322132313241325132613271328132913301331133213331334133513361337133813391340134113421343134413451346134713481349135013511352135313541355135613571358135913601361136213631364136513661367136813691370137113721373137413751376137713781379138013811382138313841385138613871388138913901391139213931394139513961397139813991400140114021403140414051406140714081409141014111412141314141415141614171418141914201421142214231424142514261427142814291430143114321433143414351436143714381439144014411442144314441445144614471448144914501451145214531454145514561457145814591460146114621463146414651466146714681469147014711472147314741475147614771478147914801481148214831484148514861487148814891490149114921493149414951496149714981499150015011502150315041505150615071508150915101511151215131514151515161517151815191520152115221523152415251526152715281529153015311532153315341535153615371538153915401541154215431544154515461547154815491550155115521553155415551556155715581559156015611562156315641565156615671568156915701571157215731574157515761577157815791580158115821583158415851586158715881589159015911592159315941595159615971598159916001601160216031604160516061607160816091610161116121613161416151616161716181619162016211622162316241625162616271628162916301631163216331634163516361637163816391640164116421643164416451646164716481649165016511652165316541655165616571658165916601661166216631664166516661667166816691670167116721673167416751676167716781679168016811682168316841685168616871688168916901691169216931694169516961697169816991700170117021703170417051706170717081709171017111712171317141715171617171718171917201721172217231724172517261727172817291730173117321733173417351736173717381739174017411742174317441745174617471748174917501751175217531754175517561757175817591760176117621763176417651766176717681769177017711772177317741775177617771778177917801781178217831784178517861787178817891790179117921793179417951796179717981799180018011802180318041805180618071808180918101811181218131814181518161817181818191820182118221823182418251826182718281829183018311832183318341835183618371838183918401841184218431844184518461847184818491850185118521853185418551856185718581859186018611862186318641865186618671868186918701871187218731874187518761877
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from '@deepseek-ai/cordis'
  3. import { createUserMessage, ToolCallId } from '@deepseek-ai/dsh-llm'
  4. import { createScope } from '@deepseek-ai/dsh-scope'
  5. import type { Scope } from '@deepseek-ai/dsh-scope'
  6. import SystemPrompt, { FIRST_PARTY_SECTION_ORDER } from '@deepseek-ai/dsh-system-prompt'
  7. import { CodeRuntime } from '@deepseek-ai/dsh-code-runtime'
  8. import type { CodeRunRequest, CodeRunResult } from '@deepseek-ai/dsh-code-runtime'
  9. import ToolRuntime, { CodeRunFailedError, RUN_CODE_NAME, TOOL_ABORTED_BEFORE_DISPATCH, defineContentToolFixture, defineTool } from '@deepseek-ai/dsh-tools'
  10. import type { Config, JsonSchemaNode, PostToolDecision, ToolExecutionResult } from '@deepseek-ai/dsh-tools'
  11. import type { Agent } from '@deepseek-ai/dsh-agent'
  12. import { Session, SessionId } from '@deepseek-ai/dsh-session'
  13. import type { JsonValue, SessionEventMap } from '@deepseek-ai/dsh-session'
  14. const testToolSignal = new AbortController().signal
  15. /**
  16. * Code Mode unit tier (per the Agent Note's plan): provider contribution per mode,
  17. * misconfiguration rejections, the run_code dispatch bridge (serialization,
  18. * abort, JSON normalization, error mapping, events, quiescence), and HMR
  19. * safety — all against an in-repo fake runtime, exactly the
  20. * Service Definition / Service Provider / Consumer roles the seam promises.
  21. */
  22. /** A scriptable in-repo CodeRuntime: each test sets `behavior` to drive the bindings however it needs. */
  23. class FakeRuntime extends CodeRuntime {
  24. readonly language: string
  25. readonly isolation = 'fake'
  26. behavior: (request: CodeRunRequest) => Promise<CodeRunResult> = () => Promise.resolve({ logs: [] })
  27. lastRequest?: CodeRunRequest
  28. constructor(ctx: Context, config: { language?: string } = {}) {
  29. super(ctx)
  30. this.language = config.language ?? 'typescript'
  31. }
  32. run(request: CodeRunRequest): Promise<CodeRunResult> {
  33. this.lastRequest = request
  34. return this.behavior(request)
  35. }
  36. }
  37. interface SetupOptions {
  38. mode?: Config['mode']
  39. maxParallelSubCalls?: number
  40. runtime?: false | { language?: string }
  41. toolOrder?: string[]
  42. }
  43. async function setup(options: SetupOptions = {}) {
  44. const ctx = new Context()
  45. await ctx.plugin(SystemPrompt, { ...options.toolOrder ? { toolOrder: options.toolOrder } : {} })
  46. await ctx.plugin(ToolRuntime, { mode: options.mode ?? 'code', ...options.maxParallelSubCalls !== undefined ? { maxParallelSubCalls: options.maxParallelSubCalls } : {} })
  47. let runtime: FakeRuntime | undefined
  48. if (options.runtime !== false) {
  49. await ctx.plugin(FakeRuntime, options.runtime ?? {})
  50. runtime = ctx.codeRuntime as FakeRuntime
  51. }
  52. return { ctx, tools: ctx.tools, systemPrompt: ctx.systemPrompt, runtime: runtime! }
  53. }
  54. /** Mint an agent scope configured like production that can register scoped tool policy. */
  55. async function mintAgentScope(ctx: Context, name = 'scoped'): Promise<{ scope: Scope; agent: Agent }> {
  56. const agent = { id: SessionId(name) } as Agent
  57. let scope!: Scope
  58. await ctx.plugin(Object.assign((inner: Context) => { scope = createScope(inner, agent) },
  59. { inject: ['tools', 'systemPrompt'] }))
  60. return { scope, agent }
  61. }
  62. /** Register a trivial echo tool; returns the calls it received. */
  63. function registerEcho(ctx: Context, name = 'echo'): unknown[] {
  64. const calls: unknown[] = []
  65. ctx.tools.register(defineTool({
  66. name,
  67. description: `Echo tool ${name}.`,
  68. parameters: { value: { type: 'string', required: true } },
  69. output: {
  70. schema: { type: 'string' },
  71. render: (_args, value) => [{ type: 'text', text: value }],
  72. },
  73. execute(args) {
  74. calls.push(args)
  75. return Promise.resolve(`${name}:${args.value}`)
  76. },
  77. }))
  78. return calls
  79. }
  80. /** A structural fake of the owning agent: captures session appends. */
  81. function fakeAgent(): { agent: Agent; events: { type: string; data: unknown }[] } {
  82. const events: { type: string; data: unknown }[] = []
  83. const agent = {
  84. session: {
  85. header: { cwd: '/workspace' },
  86. append: (type: string, data: unknown) => { events.push({ type, data }) },
  87. },
  88. } as unknown as Agent
  89. return { agent, events }
  90. }
  91. /** Dispatch run_code through the registry pipeline, as the loop would. */
  92. async function runCode(
  93. ctx: Context,
  94. code: string,
  95. extras: { agent?: Agent; signal?: AbortSignal; description?: string } = {},
  96. ): Promise<ToolExecutionResult> {
  97. return ctx.tools.execute({
  98. signal: testToolSignal,
  99. callId: ToolCallId('call-1'),
  100. name: RUN_CODE_NAME,
  101. arguments: { code, description: extras.description ?? 'Run the test program' },
  102. ...extras.agent ? { agent: extras.agent } : {},
  103. ...extras.signal ? { signal: extras.signal } : {},
  104. })
  105. }
  106. describe('mode-aware wire contribution', () => {
  107. it("mode 'native' contributes every schema, no run_code, no SDK section — and needs no runtime", async () => {
  108. const { ctx, systemPrompt } = await setup({ mode: 'native', runtime: false })
  109. registerEcho(ctx)
  110. const assembly = await systemPrompt.assemble()
  111. expect(assembly.tools.map(tool => tool.name)).toEqual(['echo'])
  112. expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
  113. })
  114. it("mode 'code' contributes exactly [run_code] plus the SDK section declaring the other tools", async () => {
  115. const { ctx, systemPrompt } = await setup({ mode: 'code' })
  116. registerEcho(ctx)
  117. const assembly = await systemPrompt.assemble()
  118. expect(assembly.tools.map(tool => tool.name)).toEqual([RUN_CODE_NAME])
  119. const sdk = assembly.sections.find(section => section.name === 'tools:sdk')
  120. expect(sdk?.text).toContain('declare const tools: {')
  121. expect(sdk?.text).toContain('echo: {')
  122. expect(sdk?.text).not.toContain('run_code:')
  123. expect(sdk?.text).not.toContain('tools.bash(')
  124. })
  125. it("mode 'code' states the run_code-only rule BEFORE the per-tool guidance that names each tool", async () => {
  126. const { ctx, systemPrompt } = await setup({ mode: 'code' })
  127. registerEcho(ctx)
  128. // Stand in for a real tool's guidance section, which names its tool without
  129. // saying how it is reached.
  130. ctx.systemPrompt.section({
  131. name: 'tool:echo',
  132. order: FIRST_PARTY_SECTION_ORDER.TOOL_READ,
  133. text: 'Use the echo tool.',
  134. })
  135. const assembly = await systemPrompt.assemble()
  136. const names = assembly.sections.map(section => section.name)
  137. const rule = assembly.sections.find(section => section.name === 'tools:code-only')
  138. expect(rule?.text).toContain(`\`${RUN_CODE_NAME}\` is the only tool you can call directly`)
  139. // The rule is worthless after the guidance it qualifies.
  140. expect(names.indexOf('tools:code-only')).toBeLessThan(names.indexOf('tool:echo'))
  141. expect(names.indexOf('tools:code-only')).toBeLessThan(names.indexOf('tools:sdk'))
  142. })
  143. it("mode 'both' omits the run_code-only rule, because native calls do execute there", async () => {
  144. const { ctx, systemPrompt } = await setup({ mode: 'both' })
  145. registerEcho(ctx)
  146. const assembly = await systemPrompt.assemble()
  147. // Registered (the deployment is non-native) but empty, so the renderer
  148. // drops it: `both` executes the native call the rule would forbid.
  149. expect(assembly.sections.find(section => section.name === 'tools:code-only')?.text).toBe('')
  150. expect(assembly.tools.map(tool => tool.name)).toContain('echo')
  151. })
  152. it('projects deeply nested output schemas into the Code Mode SDK without structured-clone recursion', async () => {
  153. const { ctx, systemPrompt } = await setup({ mode: 'code' })
  154. let output: JsonSchemaNode = { type: 'string' }
  155. for (let depth = 0; depth < 5_000; depth++) {
  156. output = { oneOf: [output, { type: 'null' }] }
  157. }
  158. ctx.tools.register({
  159. name: 'deep_output',
  160. description: 'Return a deeply nested output union.',
  161. parameters: { type: 'object', properties: {} },
  162. output: {
  163. schema: output,
  164. render: (_args, value) => [{ type: 'text', text: typeof value === 'string' ? value : 'null' }],
  165. },
  166. execute() { return Promise.resolve('ok') },
  167. })
  168. const assembly = await systemPrompt.assemble()
  169. const sdk = assembly.sections.find(section => section.name === 'tools:sdk')?.text
  170. expect(sdk).toContain('deep_output: Record<string, JsonValue>;')
  171. expect(sdk).toContain('deep_output: string | null')
  172. })
  173. it.each(['code', 'both'] as const)('treats expert assembly output as authoritative in mode %s', async (mode) => {
  174. const { ctx, systemPrompt } = await setup({ mode })
  175. registerEcho(ctx)
  176. ctx.on('system-prompt/assemble', async (_assembly, _context, next) => {
  177. const assembly = await next()
  178. return {
  179. ...assembly,
  180. sections: assembly.sections.filter(section => section.name !== 'tools:sdk'),
  181. tools: assembly.tools.filter(tool => tool.name !== RUN_CODE_NAME),
  182. }
  183. }, { prepend: true })
  184. const assembly = await systemPrompt.assemble()
  185. expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
  186. expect(assembly.tools.some(tool => tool.name === RUN_CODE_NAME)).toBe(false)
  187. })
  188. it.each(['code', 'both'] as const)('lets one scope shadow the default SDK section in mode %s', async (mode) => {
  189. const { ctx, systemPrompt } = await setup({ mode })
  190. registerEcho(ctx)
  191. const { scope, agent } = await mintAgentScope(ctx)
  192. scope.ctx.systemPrompt.section({
  193. name: 'tools:sdk',
  194. order: FIRST_PARTY_SECTION_ORDER.TOOLS_SDK,
  195. text: 'SCOPED SDK',
  196. })
  197. const scoped = await systemPrompt.assemble({ scope: agent })
  198. const global = await systemPrompt.assemble()
  199. expect(scoped.sections.find(section => section.name === 'tools:sdk')?.text).toBe('SCOPED SDK')
  200. expect(global.sections.find(section => section.name === 'tools:sdk')?.text).toContain('declare const tools:')
  201. })
  202. it("mode 'both' contributes every native schema plus run_code, and the SDK section", async () => {
  203. const { ctx, systemPrompt } = await setup({ mode: 'both' })
  204. registerEcho(ctx)
  205. const assembly = await systemPrompt.assemble()
  206. expect(assembly.tools.map(tool => tool.name)).toEqual(['echo', RUN_CODE_NAME])
  207. expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(true)
  208. })
  209. it.each(['code', 'both'] as const)('keeps the run_code transport outside scoped allow-list filtering in mode %s', async (mode) => {
  210. const { ctx, systemPrompt, runtime } = await setup({ mode })
  211. registerEcho(ctx, 'echo')
  212. registerEcho(ctx, 'hidden')
  213. const { scope, agent } = await mintAgentScope(ctx)
  214. const lift = scope.ctx.tools.restrict({ allow: ['echo'] })
  215. const assembly = await systemPrompt.assemble({ scope: agent })
  216. expect(assembly.tools.map(tool => tool.name)).toEqual(mode === 'code'
  217. ? [RUN_CODE_NAME]
  218. : ['echo', RUN_CODE_NAME])
  219. const sdk = assembly.sections.find(section => section.name === 'tools:sdk')?.text
  220. expect(sdk).toContain('echo: {')
  221. expect(sdk).not.toContain('hidden:')
  222. runtime.behavior = request => Promise.resolve({
  223. logs: [],
  224. value: Object.keys(request.bindings[0]!.functions).sort().join(','),
  225. })
  226. const result = await runCode(ctx, 'return Object.keys(tools)', { agent })
  227. expect(result.isError).toBe(false)
  228. expect(result.content).toEqual([{ type: 'text', text: 'echo' }])
  229. lift()
  230. const unrestricted = await systemPrompt.assemble({ scope: agent })
  231. expect(unrestricted.tools.map(tool => tool.name)).toEqual(mode === 'code'
  232. ? [RUN_CODE_NAME]
  233. : ['echo', 'hidden', RUN_CODE_NAME])
  234. })
  235. it.each(['code', 'both'] as const)('keeps the run_code transport outside scoped deny-list filtering in mode %s', async (mode) => {
  236. const { ctx, systemPrompt, runtime } = await setup({ mode })
  237. registerEcho(ctx, 'denied')
  238. registerEcho(ctx, 'kept')
  239. const { scope, agent } = await mintAgentScope(ctx)
  240. scope.ctx.tools.restrict({ deny: ['denied'] })
  241. const assembly = await systemPrompt.assemble({ scope: agent })
  242. expect(assembly.tools.map(tool => tool.name)).toEqual(mode === 'code'
  243. ? [RUN_CODE_NAME]
  244. : ['kept', RUN_CODE_NAME])
  245. const sdk = assembly.sections.find(section => section.name === 'tools:sdk')?.text
  246. expect(sdk).not.toContain('denied:')
  247. expect(sdk).toContain('kept: {')
  248. runtime.behavior = request => Promise.resolve({
  249. logs: [],
  250. value: Object.keys(request.bindings[0]!.functions).sort().join(','),
  251. })
  252. const result = await runCode(ctx, 'return Object.keys(tools)', { agent })
  253. expect(result.isError).toBe(false)
  254. expect(result.content).toEqual([{ type: 'text', text: 'kept' }])
  255. })
  256. it.each(['code', 'both'] as const)('reserves run_code against scoped shadows and explicit restrictions in mode %s', async (mode) => {
  257. const { ctx, systemPrompt } = await setup({ mode })
  258. const { scope, agent } = await mintAgentScope(ctx)
  259. const impostor = defineContentToolFixture({
  260. name: RUN_CODE_NAME,
  261. description: 'Scoped impostor.',
  262. parameters: {},
  263. execute: () => Promise.resolve([{ type: 'text' as const, text: 'impostor' }]),
  264. })
  265. expect(() => scope.ctx.tools.register(impostor)).toThrow(/reserved for the Code Mode presentation transport/)
  266. expect(() => ctx.tools.register(impostor)).toThrow(/reserved for the Code Mode presentation transport/)
  267. expect(() => scope.ctx.tools.restrict({ allow: [RUN_CODE_NAME] })).toThrow(/cannot name reserved Code Mode presentation transport/)
  268. expect(() => scope.ctx.tools.restrict({ deny: [RUN_CODE_NAME] })).toThrow(/cannot name reserved Code Mode presentation transport/)
  269. scope.ctx.systemPrompt.section({
  270. name: 'scoped-note',
  271. order: FIRST_PARTY_SECTION_ORDER.TOOLS_SDK - 10,
  272. text: 'safe note',
  273. })
  274. scope.ctx.tools.register(defineContentToolFixture({
  275. name: 'scoped_safe',
  276. description: 'Safe scoped tool.',
  277. parameters: {},
  278. execute: () => Promise.resolve([{ type: 'text' as const, text: 'safe' }]),
  279. }))
  280. const assembly = await systemPrompt.assemble({ scope: agent })
  281. const transports = assembly.tools.filter(tool => tool.name === RUN_CODE_NAME)
  282. expect(transports).toHaveLength(1)
  283. expect(transports[0]?.description).toContain('Execute a TypeScript program')
  284. expect(assembly.sections.find(section => section.name === 'scoped-note')?.text).toBe('safe note')
  285. expect(assembly.sections.find(section => section.name === 'tools:sdk')?.text).toContain('scoped_safe:')
  286. expect(ctx.tools.get(RUN_CODE_NAME, agent)).toBe(ctx.tools.get(RUN_CODE_NAME))
  287. const result = await runCode(ctx, 'return 1', { agent })
  288. expect(result.content).toEqual([{ type: 'text', text: '(run_code completed with no output)' }])
  289. })
  290. it.each(['code', 'both'] as const)('keeps run_code in the toolOrder universe without exposing it as a restriction target in mode %s', async (mode) => {
  291. const { ctx, systemPrompt } = await setup({
  292. mode,
  293. toolOrder: [RUN_CODE_NAME, '<unlisted-tools>'],
  294. })
  295. registerEcho(ctx)
  296. const { agent } = await mintAgentScope(ctx)
  297. const assembly = await systemPrompt.assemble({ scope: agent })
  298. expect(assembly.tools.map(tool => tool.name)).toEqual(mode === 'code'
  299. ? [RUN_CODE_NAME]
  300. : [RUN_CODE_NAME, 'echo'])
  301. })
  302. it("never exposes run_code to programs, even under mode 'both' (no recursive dispatch path)", async () => {
  303. const { ctx, runtime } = await setup({ mode: 'both' })
  304. registerEcho(ctx)
  305. runtime.behavior = (request) => {
  306. expect(request.bindings[0]!.errorClass).toEqual({
  307. name: 'ToolCallError',
  308. memberNameProperty: 'toolName',
  309. })
  310. const functions = request.bindings[0]!.functions
  311. return Promise.resolve({
  312. logs: [],
  313. value: JSON.stringify({
  314. names: Object.keys(functions).sort(),
  315. // Own-property AND prototype-chain reads both come back empty —
  316. // there is no handle a program could re-enter run_code through.
  317. runCode: String(functions[RUN_CODE_NAME]),
  318. }),
  319. })
  320. }
  321. const result = await runCode(ctx, 'program')
  322. expect(result.isError).toBe(false)
  323. expect(JSON.parse((result.content[0] as { text: string }).text)).toEqual({ names: ['echo'], runCode: 'undefined' })
  324. })
  325. it('renders byte-identical SDK text across consecutive assemblies of an unchanged tool set', async () => {
  326. const { ctx, systemPrompt } = await setup({ mode: 'code' })
  327. registerEcho(ctx)
  328. const first = await systemPrompt.assemble()
  329. const second = await systemPrompt.assemble()
  330. const text = (assembly: typeof first) => assembly.sections.find(section => section.name === 'tools:sdk')?.text
  331. expect(text(first)).toBe(text(second))
  332. })
  333. it('rejects every assembly when a non-native mode has no code runtime', async () => {
  334. const { systemPrompt } = await setup({ mode: 'code', runtime: false })
  335. await expect(systemPrompt.assemble()).rejects.toThrow(/requires a code runtime/)
  336. })
  337. it('rejects every assembly when the runtime language has no registered SDK renderer', async () => {
  338. const { systemPrompt } = await setup({ mode: 'code', runtime: { language: 'ruby' } })
  339. await expect(systemPrompt.assemble()).rejects.toThrow(/no SDK renderer registered for runtime language "ruby"/)
  340. })
  341. it('assembles under a python runtime by picking the Python SDK renderer', async () => {
  342. const { ctx, systemPrompt } = await setup({ mode: 'code', runtime: { language: 'python' } })
  343. registerEcho(ctx)
  344. const assembly = await systemPrompt.assemble()
  345. const sdk = assembly.sections.find(section => section.name === 'tools:sdk')
  346. expect(sdk?.text).toContain('class Tools(Protocol):')
  347. expect(sdk?.text).toContain('async def echo(self, args:')
  348. expect(sdk?.text).toContain('top-level `await`')
  349. })
  350. it("assembles under a python runtime in mode 'both' as well, SDK and schema together", async () => {
  351. // `both` reaches the same wireSchemas/requireCodeRuntime/SDK-section code
  352. // as `code`, so this pins the mode-by-language matrix rather than a
  353. // separate path — including that the `wireSchemas` projection behind
  354. // `assembly.tools` picks the Python flavor under `both` instead of hitting
  355. // the flavor-table guard.
  356. const { ctx, systemPrompt } = await setup({ mode: 'both', runtime: { language: 'python' } })
  357. registerEcho(ctx)
  358. const assembly = await systemPrompt.assemble()
  359. expect(assembly.sections.find(section => section.name === 'tools:sdk')?.text).toContain('class Tools(Protocol):')
  360. const runCodeSchema = assembly.tools.find(tool => tool.name === RUN_CODE_NAME)
  361. expect(runCodeSchema?.description).toContain('Execute a Python program')
  362. // `both` keeps the native tools alongside run_code; `code` does not.
  363. expect(assembly.tools.map(tool => tool.name)).toContain('echo')
  364. })
  365. it('emits a TypeScript-flavored run_code schema under a typescript runtime', async () => {
  366. const { ctx, systemPrompt } = await setup({ mode: 'code', runtime: { language: 'typescript' } })
  367. registerEcho(ctx)
  368. const assembly = await systemPrompt.assemble()
  369. const runCodeSchema = assembly.tools.find(tool => tool.name === RUN_CODE_NAME)
  370. expect(runCodeSchema?.description).toContain('Execute a TypeScript program')
  371. expect(runCodeSchema?.description).toContain('BODY of an')
  372. // Both required arguments are named here, not only in the parameter
  373. // schema: prose that describes the call as "pass the program" is what
  374. // leads a model to emit `{code}` alone and fail INVALID_ARGS.
  375. expect(runCodeSchema?.description).toContain('`description`')
  376. const codeParam = (runCodeSchema?.parameters as { properties: { code: { description: string } } }).properties.code
  377. expect(codeParam.description).toBe('The program: the body of an async TypeScript function.')
  378. })
  379. it('emits a Python-flavored run_code schema under a python runtime (matches the SDK language)', async () => {
  380. const { ctx, systemPrompt } = await setup({ mode: 'code', runtime: { language: 'python' } })
  381. registerEcho(ctx)
  382. const assembly = await systemPrompt.assemble()
  383. const runCodeSchema = assembly.tools.find(tool => tool.name === RUN_CODE_NAME)
  384. expect(runCodeSchema?.description).toContain('Execute a Python program')
  385. expect(runCodeSchema?.description).toContain('`return <value>`')
  386. expect(runCodeSchema?.description).toContain('`description`')
  387. expect(runCodeSchema?.description).not.toContain('TypeScript')
  388. const codeParam = (runCodeSchema?.parameters as { properties: { code: { description: string } } }).properties.code
  389. expect(codeParam.description).toBe('The program: the body of an async Python function.')
  390. })
  391. it('resolves the run_code schema flavor lazily and fails loud on a language absent from the flavor table', async () => {
  392. // The flavor getter reads the runtime directly (peekRuntime), so it — not
  393. // requireCodeRuntime — owns the flavor-table guard. Keeping
  394. // RUN_CODE_FLAVORS in step with SDK_RENDERERS is the compiler's job (both
  395. // are `satisfies`-checked against CodeSdkLanguage), so what the guard
  396. // covers is a mounted runtime naming a language absent from both tables,
  397. // which throws when the schema is projected. Assembly's
  398. // requireCodeRuntime rejects such a language earlier; this reaches the
  399. // guard on its own.
  400. const { ctx } = await setup({ mode: 'code', runtime: { language: 'ruby' } })
  401. const definition = ctx.tools.get(RUN_CODE_NAME)
  402. // Names the known languages, symmetric with the SDK_RENDERERS guard: this
  403. // is the reachable rejection, so it must be at least as diagnosable.
  404. expect(() => definition?.description)
  405. .toThrow(/no run_code schema flavor registered for runtime language "ruby" \(known: "typescript", "python"\)/)
  406. })
  407. it('degrades the run_code flavor to TypeScript when no runtime is mounted', async () => {
  408. // Any reader of the definition without a mounted runtime uses this fallback; the
  409. // shipped one is the tool-catalog generator, which boots the registry under
  410. // `mode: code` and reads run_code's schema WITHOUT a runtime. peekRuntime
  411. // returns undefined there, so the flavor getter degrades to the TS default
  412. // rather than throwing. None of those readers feeds a model: assembly goes
  413. // through wireSchemas, which requires a runtime first.
  414. const { ctx } = await setup({ mode: 'code', runtime: false })
  415. const definition = ctx.tools.get(RUN_CODE_NAME)
  416. expect(definition?.description).toContain('Execute a TypeScript program')
  417. const params = definition?.parameters as { properties: { code: { description: string } } }
  418. expect(params.properties.code.description).toBe('The program: the body of an async TypeScript function.')
  419. })
  420. it("rejects the assembly when toolOrder names a native tool that mode 'code' no longer contributes", async () => {
  421. const { ctx, systemPrompt } = await setup({ mode: 'code', toolOrder: ['echo', '<unlisted-tools>'] })
  422. registerEcho(ctx)
  423. await expect(systemPrompt.assemble()).rejects.toThrow(/toolOrder lists unregistered tool "echo"/)
  424. })
  425. it('removes run_code and the SDK section when the registry fiber disposes (HMR safety)', async () => {
  426. const ctx = new Context()
  427. await ctx.plugin(SystemPrompt, {})
  428. await ctx.plugin(FakeRuntime, {})
  429. const fiber = await ctx.plugin(ToolRuntime, { mode: 'code' })
  430. expect(ctx.tools.get(RUN_CODE_NAME)).toBeDefined()
  431. await fiber.dispose()
  432. const assembly = await ctx.systemPrompt.assemble()
  433. expect(assembly.tools).toEqual([])
  434. expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
  435. })
  436. })
  437. describe('the sub-dispatch scheduler (native concurrency contract)', () => {
  438. /** Register a tool whose calls resolve only when the test releases them; returns live-call telemetry. */
  439. function registerGated(ctx: Context, name: string, concurrencySafe: boolean) {
  440. const gates: (() => void)[] = []
  441. let live = 0
  442. let peak = 0
  443. const order: string[] = []
  444. ctx.tools.register(defineTool({
  445. name,
  446. description: `Gated tool ${name}.`,
  447. parameters: { id: { type: 'string', required: true } },
  448. output: {
  449. schema: { type: 'string' },
  450. render: (_args, value) => [{ type: 'text', text: value }],
  451. },
  452. ...concurrencySafe ? { isConcurrencySafe: () => true } : {},
  453. async execute(args, exec) {
  454. order.push(`start:${args.id}`)
  455. live++
  456. peak = Math.max(peak, live)
  457. // Abort-observing like a real tool: the run-scoped abort releases the
  458. // gate so the bridge's drain reaches quiescence.
  459. await new Promise<void>((release) => {
  460. gates.push(release)
  461. exec.signal.addEventListener('abort', () => { release() }, { once: true })
  462. })
  463. live--
  464. order.push(`end:${args.id}`)
  465. return `${name}:${args.id}`
  466. },
  467. }))
  468. const release = (): void => { gates.shift()?.() }
  469. const releaseAll = (): void => { while (gates.length > 0) gates.shift()!() }
  470. return { order, release, releaseAll, peakLive: () => peak, pending: () => gates.length }
  471. }
  472. it('overlaps concurrency-safe calls under Promise.all and logs a start event per dispatch', async () => {
  473. const { ctx, runtime } = await setup({ mode: 'code' })
  474. const gated = registerGated(ctx, 'safe_read', true)
  475. const { agent, events } = fakeAgent()
  476. runtime.behavior = async (request) => {
  477. const tools = request.bindings[0]!.functions
  478. const all = Promise.all([
  479. tools.safe_read!({ id: 'a' }),
  480. tools.safe_read!({ id: 'b' }),
  481. tools.safe_read!({ id: 'c' }),
  482. ])
  483. // All three must be START-able without any completion (overlap proof).
  484. await expect.poll(() => gated.pending()).toBe(3)
  485. gated.releaseAll()
  486. return { logs: [], value: (await all).map(String).join(',') }
  487. }
  488. const result = await runCode(ctx, 'program', { agent })
  489. expect(result.isError).toBe(false)
  490. expect(gated.peakLive()).toBe(3)
  491. if (result.isError) throw new Error('expected success')
  492. expect(result.value).toMatchObject({ result: 'safe_read:a,safe_read:b,safe_read:c' })
  493. // One start per dispatch, paired with its settle by subCallId, starts in submission order.
  494. const starts = events.filter(event => event.type === 'tool/code-dispatch-start').map(event => event.data as { subCallId: string })
  495. const settles = events.filter(event => event.type === 'tool/code-dispatch').map(event => event.data as { subCallId: string })
  496. expect(starts.map(start => start.subCallId)).toEqual(['call-1:code:1', 'call-1:code:2', 'call-1:code:3'])
  497. expect(new Set(settles.map(settle => settle.subCallId))).toEqual(new Set(starts.map(start => start.subCallId)))
  498. })
  499. it('an exclusive call bars overlap: safe calls drain first, it runs alone, later calls wait', async () => {
  500. const { ctx, runtime } = await setup({ mode: 'code' })
  501. const safe = registerGated(ctx, 'safe_read', true)
  502. const unsafe = registerGated(ctx, 'writer', false)
  503. runtime.behavior = async (request) => {
  504. const tools = request.bindings[0]!.functions
  505. const reads = [tools.safe_read!({ id: 'r1' }), tools.safe_read!({ id: 'r2' })]
  506. const write = tools.writer!({ id: 'w' })
  507. const tail = tools.safe_read!({ id: 'r3' })
  508. await expect.poll(() => safe.pending()).toBe(2)
  509. // The exclusive call must NOT have started while the pool is live.
  510. expect(unsafe.pending()).toBe(0)
  511. safe.releaseAll()
  512. await expect.poll(() => unsafe.pending()).toBe(1)
  513. // The trailing safe call must NOT start while the exclusive one runs.
  514. expect(safe.pending()).toBe(0)
  515. unsafe.release()
  516. await expect.poll(() => safe.pending()).toBe(1)
  517. safe.releaseAll()
  518. await Promise.all([...reads, write, tail])
  519. return { logs: [], value: 'ordered' }
  520. }
  521. const result = await runCode(ctx, 'program')
  522. expect(result.isError).toBe(false)
  523. expect(safe.order.slice(0, 2)).toEqual(['start:r1', 'start:r2'])
  524. expect(unsafe.order).toEqual(['start:w', 'end:w'])
  525. // r3 started only after w ended.
  526. expect(safe.order.indexOf('start:r3')).toBeGreaterThan(safe.order.indexOf('end:r1'))
  527. })
  528. it('maxParallelSubCalls caps the overlap window', async () => {
  529. const { ctx, runtime } = await setup({ mode: 'code', maxParallelSubCalls: 2 })
  530. const gated = registerGated(ctx, 'safe_read', true)
  531. runtime.behavior = async (request) => {
  532. const tools = request.bindings[0]!.functions
  533. const all = Promise.all([
  534. tools.safe_read!({ id: 'a' }),
  535. tools.safe_read!({ id: 'b' }),
  536. tools.safe_read!({ id: 'c' }),
  537. ])
  538. await expect.poll(() => gated.pending()).toBe(2)
  539. // The third call waits for a slot.
  540. expect(gated.pending()).toBe(2)
  541. gated.release()
  542. await expect.poll(() => gated.pending()).toBe(2)
  543. gated.releaseAll()
  544. await all
  545. return { logs: [], value: 'capped' }
  546. }
  547. const result = await runCode(ctx, 'program')
  548. if (result.isError) console.error('CAP-FAIL:', (result.content[0] as { text: string }).text)
  549. expect(result.isError).toBe(false)
  550. expect(gated.peakLive()).toBe(2)
  551. })
  552. it('a tool unregistered between binding enumeration and dispatch fails as unknown tool', async () => {
  553. const { ctx, runtime } = await setup({ mode: 'code' })
  554. const calls: unknown[] = []
  555. const dispose = ctx.tools.register(defineTool({
  556. name: 'ephemeral',
  557. description: 'Unregistered between binding enumeration and dispatch.',
  558. parameters: {},
  559. output: {
  560. schema: { type: 'string' },
  561. render: (_args, value) => [{ type: 'text', text: value }],
  562. },
  563. execute() {
  564. calls.push('ran')
  565. return Promise.resolve('ok')
  566. },
  567. }))
  568. runtime.behavior = async (request) => {
  569. // The binding exists (enumerated at run start); the registry mutation
  570. // makes prepare resolve UNKNOWN_TOOL as a final-result, which commits
  571. // through scheduler.finish (no post-execute).
  572. dispose()
  573. const message = await request.bindings[0]!.functions.ephemeral!({})
  574. .then(() => 'resolved', (error: unknown) => error instanceof Error ? error.message : String(error))
  575. return { logs: [], value: message }
  576. }
  577. const result = await runCode(ctx, 'program')
  578. expect(result.isError).toBe(false)
  579. if (result.isError) throw new Error('expected success')
  580. expect(result.value).toMatchObject({ result: 'unknown tool "ephemeral"' })
  581. expect(calls).toEqual([])
  582. })
  583. it('ordered pre-execute never overlaps: a slow policy on one call delays the next start', async () => {
  584. const { ctx, runtime } = await setup({ mode: 'code' })
  585. const gated = registerGated(ctx, 'safe_read', true)
  586. const stages: string[] = []
  587. let releaseGate: (() => void) | undefined
  588. ctx.on('tools/pre-execute', async (preExec, next) => {
  589. if (preExec.name !== 'safe_read') return next()
  590. stages.push(`pre-enter:${String(preExec.callId)}`)
  591. if (releaseGate === undefined) {
  592. // The FIRST call's policy awaits an asynchronous decision.
  593. await new Promise<void>((resolve) => { releaseGate = resolve })
  594. }
  595. stages.push(`pre-exit:${String(preExec.callId)}`)
  596. return next()
  597. })
  598. runtime.behavior = async (request) => {
  599. const tools = request.bindings[0]!.functions
  600. const all = Promise.all([tools.safe_read!({ id: 'a' }), tools.safe_read!({ id: 'b' })])
  601. // Both submissions are in; the second pre-execute must NOT have entered
  602. // while the first is still awaiting its policy decision.
  603. await expect.poll(() => stages.length).toBeGreaterThanOrEqual(1)
  604. expect(stages).toEqual(['pre-enter:call-1:code:1'])
  605. releaseGate!()
  606. await expect.poll(() => gated.pending()).toBe(2)
  607. gated.releaseAll()
  608. await all
  609. return { logs: [], value: 'ordered-prepare' }
  610. }
  611. const result = await runCode(ctx, 'program')
  612. expect(result.isError).toBe(false)
  613. expect(stages).toEqual([
  614. 'pre-enter:call-1:code:1', 'pre-exit:call-1:code:1',
  615. 'pre-enter:call-1:code:2', 'pre-exit:call-1:code:2',
  616. ])
  617. })
  618. it('an exclusive call holds its barrier through post-execute: the next start waits for the commit', async () => {
  619. const { ctx, runtime } = await setup({ mode: 'code' })
  620. const writer = registerGated(ctx, 'writer', false)
  621. const reader = registerGated(ctx, 'safe_read', true)
  622. const stages: string[] = []
  623. let releasePost: (() => void) | undefined
  624. ctx.on('tools/post-execute', async (postExec, _result, next): Promise<PostToolDecision> => {
  625. if (postExec.name === 'writer') {
  626. stages.push('post-enter:writer')
  627. await new Promise<void>((resolve) => { releasePost = resolve })
  628. stages.push('post-exit:writer')
  629. }
  630. return next()
  631. })
  632. runtime.behavior = async (request) => {
  633. const tools = request.bindings[0]!.functions
  634. const w = tools.writer!({ id: 'w' })
  635. const r = tools.safe_read!({ id: 'r' })
  636. await expect.poll(() => writer.pending()).toBe(1)
  637. writer.release()
  638. // The writer's body is done and its async post-execute is running; the
  639. // parallel read must not have STARTED (no pre/body) while the exclusive
  640. // call's pipeline is still open.
  641. await expect.poll(() => stages).toContain('post-enter:writer')
  642. expect(reader.pending()).toBe(0)
  643. releasePost!()
  644. await w
  645. await expect.poll(() => reader.pending()).toBe(1)
  646. reader.releaseAll()
  647. await r
  648. return { logs: [], value: 'barrier-through-commit' }
  649. }
  650. const result = await runCode(ctx, 'program')
  651. expect(result.isError).toBe(false)
  652. expect(stages).toEqual(['post-enter:writer', 'post-exit:writer'])
  653. })
  654. it('run settlement drains a commit already in progress: the settle event is appended inside the turn', async () => {
  655. const { ctx, runtime } = await setup({ mode: 'code' })
  656. const gated = registerGated(ctx, 'safe_read', true)
  657. const { agent, events } = fakeAgent()
  658. let releasePost: (() => void) | undefined
  659. ctx.on('tools/post-execute', async (postExec, _result, next): Promise<PostToolDecision> => {
  660. if (postExec.name === 'safe_read') {
  661. await new Promise<void>((resolve) => { releasePost = resolve })
  662. }
  663. return next()
  664. })
  665. runtime.behavior = async (request) => {
  666. // Fire-and-forget: the program returns while the sub-call's async
  667. // post-execute commit is mid-flight.
  668. request.bindings[0]!.functions.safe_read!({ id: 'a' }).catch(() => 'run-over')
  669. await expect.poll(() => gated.pending()).toBe(1)
  670. gated.release()
  671. await expect.poll(() => releasePost !== undefined).toBe(true)
  672. queueMicrotask(() => { releasePost!() })
  673. return { logs: [], value: 'returned-early' }
  674. }
  675. const result = await runCode(ctx, 'program', { agent })
  676. expect(result.isError).toBe(false)
  677. // The drain awaited the in-progress commit: the settle event exists and
  678. // preceded the run_code turn closing (all appends happen inside
  679. // execute()). The run's settlement aborted the sub-call's signal while
  680. // its post-execute was mid-flight, so the native cancellation contract
  681. // replaces the successful outcome with the aborted result — the event is
  682. // still durable and in-turn, which is the invariant under test.
  683. const settles = events.filter(event => event.type === 'tool/code-dispatch')
  684. expect(settles).toHaveLength(1)
  685. expect(settles[0]?.data).toMatchObject({ name: 'safe_read', isError: true })
  686. })
  687. it('post-execute and context commitment stay in submission order under out-of-order completion', async () => {
  688. const { ctx, runtime } = await setup({ mode: 'code' })
  689. const gated = registerGated(ctx, 'safe_read', true)
  690. const postOrder: string[] = []
  691. ctx.on('tools/post-execute', async (postExec, _result, next): Promise<PostToolDecision> => {
  692. if (postExec.name === 'safe_read') {
  693. postOrder.push(String(postExec.callId))
  694. return {
  695. kind: 'accept' as const,
  696. additionalContexts: [createUserMessage({
  697. content: [{ type: 'text' as const, text: `ctx:${String(postExec.callId)}` }],
  698. source: { kind: 'plugin' as const, plugin: 'order-probe' },
  699. })],
  700. }
  701. }
  702. return next()
  703. })
  704. runtime.behavior = async (request) => {
  705. const tools = request.bindings[0]!.functions
  706. const all = Promise.all([tools.safe_read!({ id: 'a' }), tools.safe_read!({ id: 'b' })])
  707. await expect.poll(() => gated.pending()).toBe(2)
  708. // Complete b FIRST (out of submission order), then a.
  709. gated.release() // releases a (FIFO gate) — invert: release twice reversed is not possible;
  710. gated.releaseAll()
  711. await all
  712. return { logs: [], value: 'ordered-commit' }
  713. }
  714. const result = await runCode(ctx, 'program')
  715. expect(result.isError).toBe(false)
  716. // Post-execute observed submission order regardless of completion interleave.
  717. expect(postOrder).toEqual(['call-1:code:1', 'call-1:code:2'])
  718. // Deferred contexts reach the outer result in the same order.
  719. expect(result.additionalContexts?.map(c => (c.content[0] as { text: string }).text))
  720. .toEqual(['ctx:call-1:code:1', 'ctx:call-1:code:2'])
  721. })
  722. it('a queued-unstarted call abandoned by run settlement logs no start event', async () => {
  723. const { ctx, runtime } = await setup({ mode: 'code' })
  724. const gated = registerGated(ctx, 'writer', false)
  725. const { agent, events } = fakeAgent()
  726. const abandoned: string[] = []
  727. runtime.behavior = async (request) => {
  728. const tools = request.bindings[0]!.functions
  729. // First exclusive call occupies the pool; the second queues unstarted.
  730. // Both rejections are captured (abandonment fires only at settlement,
  731. // AFTER this program has already failed — awaiting it here would deadlock).
  732. tools.writer!({ id: 'w1' }).catch(() => 'settled-under-abort')
  733. tools.writer!({ id: 'w2' }).catch((error: unknown) => {
  734. abandoned.push(error instanceof Error ? error.message : String(error))
  735. })
  736. await expect.poll(() => gated.pending()).toBe(1)
  737. // Fail the program while w1 is in flight and w2 is queued unstarted.
  738. throw new Error('program failed with a queued call')
  739. }
  740. const result = await runCode(ctx, 'program', { agent })
  741. expect(result.isError).toBe(true)
  742. const starts = events.filter(event => event.type === 'tool/code-dispatch-start').map(event => (event.data as { subCallId: string }).subCallId)
  743. const settles = events.filter(event => event.type === 'tool/code-dispatch').map(event => (event.data as { subCallId: string }).subCallId)
  744. // w1 started and settled under the abort; w2 never started and never
  745. // settled — no start event, no settle event, binding rejected with the
  746. // abandonment message at drain time.
  747. expect(starts).toEqual(['call-1:code:1'])
  748. expect(settles).toEqual(['call-1:code:1'])
  749. expect(abandoned).toEqual(['run_code run is over (run_code settled); writer tool call abandoned'])
  750. })
  751. })
  752. describe('the run_code dispatch bridge', () => {
  753. it('bridges tool calls, returns only the curated output, and logs one event per dispatch', async () => {
  754. const { ctx, runtime } = await setup({ mode: 'code' })
  755. const calls = registerEcho(ctx)
  756. const { agent, events } = fakeAgent()
  757. runtime.behavior = async (request) => {
  758. const tools = request.bindings[0]!.functions
  759. const first = await tools.echo!({ value: 'one' })
  760. const second = await tools.echo!({ value: 'two' })
  761. if (typeof first !== 'string' || typeof second !== 'string') throw new Error('echo returned a non-string')
  762. return { logs: [`saw ${first}`], value: second }
  763. }
  764. const result = await runCode(ctx, 'const …: string = …', { agent })
  765. expect(result.isError).toBe(false)
  766. if (result.isError) throw new Error('expected run_code success')
  767. expect(result.value).toEqual({ logs: ['saw echo:one'], result: 'echo:two' })
  768. expect(result.content).toEqual([{ type: 'text', text: 'saw echo:one\necho:two' }])
  769. expect(calls).toEqual([{ value: 'one' }, { value: 'two' }])
  770. const dispatches = events.filter(event => event.type === 'tool/code-dispatch')
  771. expect(dispatches.map(event => event.data)).toEqual([
  772. {
  773. rootCallId: 'call-1', parentCallId: 'call-1', subCallId: 'call-1:code:1', name: 'echo',
  774. arguments: { value: 'one' }, isError: false, content: [{ type: 'text', text: 'echo:one' }],
  775. },
  776. {
  777. rootCallId: 'call-1', parentCallId: 'call-1', subCallId: 'call-1:code:2', name: 'echo',
  778. arguments: { value: 'two' }, isError: false, content: [{ type: 'text', text: 'echo:two' }],
  779. },
  780. ])
  781. expect(result.meta).toBeUndefined()
  782. })
  783. it('exposes only an opaque parent token to nested result observers', async () => {
  784. const { ctx, runtime } = await setup({ mode: 'code' })
  785. registerEcho(ctx)
  786. runtime.behavior = async (request) => {
  787. await request.bindings[0]!.functions.echo!({ value: 'nested' })
  788. return { logs: [], value: 'done' }
  789. }
  790. // Freeze the nested observer's parent correlation. If that were the live
  791. // outer execution object, the timeout-style wrapper could not restore it.
  792. ctx.on('tools/execute', async (exec, next) => {
  793. if (exec.name !== RUN_CODE_NAME) return next()
  794. const previous = exec.signal
  795. exec.signal = new AbortController().signal
  796. const result = await next()
  797. exec.signal = previous
  798. return result
  799. })
  800. ctx.on('tools/result', (exec) => {
  801. if (exec.parent !== undefined) Object.freeze(exec.parent)
  802. })
  803. const result = await runCode(ctx, 'await tools.echo({ value: "nested" })')
  804. expect(result.isError).toBe(false)
  805. expect(result.content).toEqual([{ type: 'text', text: 'done' }])
  806. })
  807. it('forwards a nested terminal conclusion onto the successful run_code result', async () => {
  808. const { ctx, runtime } = await setup({ mode: 'code' })
  809. ctx.tools.register(defineTool({
  810. name: 'finalize',
  811. description: 'Terminal tool.',
  812. parameters: {},
  813. output: {
  814. schema: { type: 'string' },
  815. render: (_args, value) => [{ type: 'text', text: value }],
  816. },
  817. execute(_args, exec) {
  818. exec.concludeTurn()
  819. return Promise.resolve('done')
  820. },
  821. }))
  822. runtime.behavior = async (request) => {
  823. await request.bindings[0]!.functions.finalize!({})
  824. return { logs: [], value: 'program complete' }
  825. }
  826. const concluded = await runCode(ctx, 'await tools.finalize({})')
  827. expect(concluded.isError).toBe(false)
  828. expect(concluded.concludesTurn).toBe(true)
  829. // A policy that converts the nested success into an error strips the
  830. // marker with the result type: the recovering program cannot conclude.
  831. const veto = ctx.on('tools/post-execute', async (exec, _result, next): Promise<PostToolDecision> => {
  832. if (exec.name !== 'finalize') return next()
  833. return { kind: 'block', feedback: [{ type: 'text', text: 'terminal rejected' }] }
  834. })
  835. runtime.behavior = async (request) => {
  836. await request.bindings[0]!.functions.finalize!({}).catch(() => undefined)
  837. return { logs: [], value: 'recovered' }
  838. }
  839. const recovered = await runCode(ctx, 'await tools.finalize({}).catch(() => {})')
  840. veto()
  841. expect(recovered.isError).toBe(false)
  842. expect(recovered.concludesTurn).toBeUndefined()
  843. })
  844. it('serializes Promise.all dispatches: tool executions never overlap, in submission order', async () => {
  845. const { ctx, runtime } = await setup({ mode: 'code' })
  846. const intervals: [string, string][] = []
  847. let active = 0
  848. ctx.tools.register(defineTool({
  849. name: 'probe',
  850. description: 'Records execution overlap.',
  851. parameters: { id: { type: 'string', required: true } },
  852. output: {
  853. schema: { type: 'string' },
  854. render: (_args, value) => [{ type: 'text', text: value }],
  855. },
  856. async execute(args) {
  857. active++
  858. expect(active, 'probe executions overlapped').toBe(1)
  859. intervals.push(['enter', args.id])
  860. await new Promise(resolve => setTimeout(resolve, 20))
  861. intervals.push(['exit', args.id])
  862. active--
  863. return args.id
  864. },
  865. }))
  866. runtime.behavior = async (request) => {
  867. const tools = request.bindings[0]!.functions
  868. const values = await Promise.all([tools.probe!({ id: 'a' }), tools.probe!({ id: 'b' }), tools.probe!({ id: 'c' })])
  869. if (!values.every(value => typeof value === 'string')) throw new Error('probe returned a non-string')
  870. return { logs: [], value: values.join(',') }
  871. }
  872. const result = await runCode(ctx, 'program')
  873. expect(result.isError).toBe(false)
  874. expect(intervals).toEqual([
  875. ['enter', 'a'], ['exit', 'a'],
  876. ['enter', 'b'], ['exit', 'b'],
  877. ['enter', 'c'], ['exit', 'c'],
  878. ])
  879. expect(result.content[0]).toEqual({ type: 'text', text: 'a,b,c' })
  880. })
  881. it('rejects the program-side call when the tool errors, with the tool error text', async () => {
  882. const { ctx, runtime } = await setup({ mode: 'code' })
  883. ctx.tools.register(defineContentToolFixture({
  884. name: 'fail',
  885. description: 'Always fails.',
  886. parameters: {},
  887. execute(): Promise<never> { return Promise.reject(new Error('deliberate failure')) },
  888. }))
  889. runtime.behavior = async (request) => {
  890. try {
  891. await request.bindings[0]!.functions.fail!({})
  892. return { logs: [], value: 'unreachable' }
  893. } catch (error: unknown) {
  894. return { logs: [], value: `caught: ${error instanceof Error ? error.message : String(error)}` }
  895. }
  896. }
  897. const result = await runCode(ctx, 'program')
  898. expect(result.content[0]).toEqual({ type: 'text', text: 'caught: deliberate failure' })
  899. })
  900. it('a throwing tools/code-dispatch-log listener is contained: the original settled content is logged', async () => {
  901. const { ctx, runtime } = await setup({ mode: 'code' })
  902. registerEcho(ctx)
  903. ctx.on('tools/code-dispatch-log', () => { throw new Error('log-content listener failed') })
  904. const { agent, events } = fakeAgent()
  905. runtime.behavior = async (request) => {
  906. const value = await request.bindings[0]!.functions.echo!({ value: 'x' })
  907. return { logs: [], value: value as string }
  908. }
  909. const result = await runCode(ctx, 'program', { agent })
  910. expect(result.isError).toBe(false)
  911. const settle = events.find(event => event.type === 'tool/code-dispatch')
  912. expect(settle?.data).toMatchObject({ name: 'echo', isError: false, content: [{ type: 'text', text: 'echo:x' }] })
  913. })
  914. it('a throwing tools/pre-execute listener settles the sub-call without post-execute', async () => {
  915. const { ctx, runtime } = await setup({ mode: 'code' })
  916. const calls = registerEcho(ctx)
  917. const postExecuted: string[] = []
  918. ctx.on('tools/pre-execute', (exec, next) => {
  919. if (exec.name === 'echo') throw new Error('gate exploded')
  920. return next()
  921. })
  922. ctx.on('tools/post-execute', (exec, _result, next): Promise<PostToolDecision> => {
  923. if (exec.name === 'echo') postExecuted.push(exec.name)
  924. return next()
  925. })
  926. const { agent, events } = fakeAgent()
  927. runtime.behavior = async (request) => {
  928. const message = await request.bindings[0]!.functions.echo!({ value: 'x' })
  929. .then(() => 'resolved', (error: unknown) => error instanceof Error ? error.message : String(error))
  930. return { logs: [], value: message }
  931. }
  932. const result = await runCode(ctx, 'program', { agent })
  933. expect(result.isError).toBe(false)
  934. if (result.isError) throw new Error('expected success')
  935. expect(result.value).toMatchObject({ result: 'gate exploded' })
  936. // The pipeline failure is final: the body never ran and post-execute was
  937. // skipped, yet the settle event still carries the error outcome.
  938. expect(calls).toEqual([])
  939. expect(postExecuted).toEqual([])
  940. const settles = events.filter(event => event.type === 'tool/code-dispatch')
  941. expect(settles).toHaveLength(1)
  942. expect(settles[0]?.data).toMatchObject({ name: 'echo', isError: true })
  943. })
  944. it('a tools/pre-execute deny reaches the program as a binding rejection', async () => {
  945. const { ctx, runtime } = await setup({ mode: 'code' })
  946. registerEcho(ctx)
  947. ctx.on('tools/pre-execute', (exec, next) => {
  948. if (exec.name === 'echo') return Promise.resolve({ kind: 'deny' as const, reason: 'not on my watch' })
  949. return next()
  950. })
  951. runtime.behavior = async (request) => {
  952. try {
  953. await request.bindings[0]!.functions.echo!({ value: 'x' })
  954. return { logs: [], value: 'unreachable' }
  955. } catch (error: unknown) {
  956. return { logs: [], value: `denied: ${error instanceof Error ? error.message : String(error)}` }
  957. }
  958. }
  959. const result = await runCode(ctx, 'program')
  960. expect(result.content[0]?.type).toBe('text')
  961. expect((result.content[0] as { text: string }).text).toContain('not on my watch')
  962. })
  963. it('rejects a binding argument that is not lossless JSON, dispatching nothing', async () => {
  964. const { ctx, runtime } = await setup({ mode: 'code' })
  965. const calls = registerEcho(ctx)
  966. const { agent, events } = fakeAgent()
  967. runtime.behavior = async (request) => {
  968. try {
  969. await request.bindings[0]!.functions.echo!({ value: 'x', big: 1n })
  970. return { logs: [], value: 'unreachable' }
  971. } catch (error: unknown) {
  972. return { logs: [], value: error instanceof Error ? error.message : String(error) }
  973. }
  974. }
  975. const result = await runCode(ctx, 'program', { agent })
  976. expect((result.content[0] as { text: string }).text).toContain('lossless JSON')
  977. expect(calls).toEqual([])
  978. expect(events.filter(event => event.type === 'tool/code-dispatch')).toEqual([])
  979. })
  980. it('dispatches and logs independent snapshots of the same lossless JSON value', async () => {
  981. const { ctx, runtime } = await setup({ mode: 'code' })
  982. const calls = registerEcho(ctx)
  983. const { agent, events } = fakeAgent()
  984. runtime.behavior = async (request) => {
  985. const args = Object.assign(Object.create(null) as Record<string, unknown>, { value: 'x', nested: ['same'] })
  986. await request.bindings[0]!.functions.echo!(args)
  987. return { logs: [] }
  988. }
  989. await runCode(ctx, 'program', { agent })
  990. expect(calls).toEqual([{ value: 'x', nested: ['same'] }])
  991. const dispatch = events.find(event => event.type === 'tool/code-dispatch')?.data as SessionEventMap['tool/code-dispatch']
  992. expect(dispatch.arguments).toEqual({ value: 'x', nested: ['same'] })
  993. })
  994. it('defers sub-call additionalContexts onto the outer run_code result', async () => {
  995. const { ctx, runtime } = await setup({ mode: 'code' })
  996. registerEcho(ctx)
  997. ctx.on('tools/post-execute', (exec, _result, next): Promise<PostToolDecision> => {
  998. if (exec.name === 'echo') {
  999. return Promise.resolve({
  1000. kind: 'accept' as const,
  1001. additionalContexts: [createUserMessage({
  1002. content: [{ type: 'text' as const, text: `context for ${exec.callId}` }],
  1003. source: { kind: 'plugin' as const, plugin: 'test' },
  1004. })],
  1005. })
  1006. }
  1007. return next()
  1008. })
  1009. runtime.behavior = async (request) => {
  1010. await request.bindings[0]!.functions.echo!({ value: 'x' })
  1011. await request.bindings[0]!.functions.echo!({ value: 'y' })
  1012. return { logs: [], value: 'done' }
  1013. }
  1014. const result = await runCode(ctx, 'program')
  1015. expect(result.isError).toBe(false)
  1016. expect(result.additionalContexts).toMatchObject([
  1017. {
  1018. role: 'user',
  1019. content: [{ type: 'text', text: 'context for call-1:code:1' }],
  1020. source: { kind: 'plugin', plugin: 'test' },
  1021. },
  1022. {
  1023. role: 'user',
  1024. content: [{ type: 'text', text: 'context for call-1:code:2' }],
  1025. source: { kind: 'plugin', plugin: 'test' },
  1026. },
  1027. ])
  1028. })
  1029. it('defers image-bearing final sub-call content onto the outer run_code result', async () => {
  1030. const { ctx, runtime } = await setup({ mode: 'code' })
  1031. ctx.tools.register(defineContentToolFixture({
  1032. name: 'image_result',
  1033. description: 'Return one durable image.',
  1034. parameters: {},
  1035. execute: () => Promise.resolve([
  1036. { type: 'text', text: 'image result' },
  1037. {
  1038. type: 'image',
  1039. attachment: {
  1040. attachmentId: 'sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa' as never,
  1041. mediaType: 'image/png', bytes: 1, width: 1, height: 1,
  1042. },
  1043. },
  1044. ]),
  1045. }))
  1046. runtime.behavior = async (request) => {
  1047. await request.bindings[0]!.functions.image_result!({})
  1048. return { logs: [], value: 'done' }
  1049. }
  1050. const result = await runCode(ctx, 'program')
  1051. expect(result.additionalContexts).toMatchObject([{
  1052. role: 'user',
  1053. source: { kind: 'plugin', plugin: 'tools-code-mode' },
  1054. content: [
  1055. { type: 'text', text: 'image result' },
  1056. { type: 'image', attachment: { mediaType: 'image/png', bytes: 1, width: 1, height: 1 } },
  1057. ],
  1058. }])
  1059. })
  1060. it('does not defer images removed by a nested post-execute decision', async () => {
  1061. for (const decision of ['block', 'replace'] as const) {
  1062. const { ctx, runtime } = await setup({ mode: 'code' })
  1063. ctx.tools.register(defineContentToolFixture({
  1064. name: 'image_result',
  1065. description: 'Return one durable image.',
  1066. parameters: {},
  1067. execute: () => Promise.resolve([{
  1068. type: 'image',
  1069. attachment: {
  1070. attachmentId: 'sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb' as never,
  1071. mediaType: 'image/png', bytes: 1, width: 1, height: 1,
  1072. },
  1073. }]),
  1074. }))
  1075. ctx.on('tools/post-execute', (exec, _result, next): Promise<PostToolDecision> => {
  1076. if (exec.name !== 'image_result') return next()
  1077. return Promise.resolve(decision === 'block'
  1078. ? { kind: 'block', feedback: [{ type: 'text', text: 'blocked' }] }
  1079. : { kind: 'accept', content: [{ type: 'text', text: 'replaced' }] })
  1080. })
  1081. runtime.behavior = async (request) => {
  1082. await request.bindings[0]!.functions.image_result!({}).catch(() => undefined)
  1083. return { logs: [], value: 'done' }
  1084. }
  1085. const result = await runCode(ctx, 'program')
  1086. expect(result.additionalContexts).toBeUndefined()
  1087. await ctx.fiber.dispose()
  1088. }
  1089. })
  1090. it('keeps sub-call contexts when run_code fails after the nested dispatch', async () => {
  1091. const { ctx, runtime } = await setup({ mode: 'both' })
  1092. registerEcho(ctx)
  1093. ctx.on('tools/post-execute', (exec, _result, next): Promise<PostToolDecision> => {
  1094. if (exec.name !== 'echo') return next()
  1095. return Promise.resolve({
  1096. kind: 'accept',
  1097. additionalContexts: [createUserMessage({
  1098. content: [{ type: 'text', text: 'nested context' }],
  1099. source: { kind: 'plugin', plugin: 'test' },
  1100. })],
  1101. })
  1102. })
  1103. runtime.behavior = async (request) => {
  1104. await request.bindings[0]!.functions.echo!({ value: 'x' })
  1105. return { logs: [], error: { kind: 'exception', message: 'program failed later' } }
  1106. }
  1107. const result = await runCode(ctx, 'program')
  1108. expect(result.isError).toBe(true)
  1109. expect(result.additionalContexts).toEqual([{
  1110. id: expect.any(String) as unknown,
  1111. role: 'user',
  1112. content: [{ type: 'text', text: 'nested context' }],
  1113. source: { kind: 'plugin', plugin: 'test' },
  1114. }])
  1115. })
  1116. it('converts a failed run into a structured isError result carrying kind, message, and captured logs', async () => {
  1117. const { ctx, runtime } = await setup({ mode: 'code' })
  1118. runtime.behavior = () => Promise.resolve({
  1119. logs: ['got this far'],
  1120. error: { kind: 'timeout', message: 'compute budget exhausted (300ms busy)' },
  1121. })
  1122. const result = await runCode(ctx, 'program')
  1123. expect(result.isError).toBe(true)
  1124. expect(result.error).toMatchObject({ info: { name: 'CodeRunFailedError', code: 'CODE_RUN_FAILED' } })
  1125. const text = (result.content[0] as { text: string }).text
  1126. expect(text).toContain('code run failed (timeout)')
  1127. expect(text).toContain('compute budget exhausted')
  1128. expect(text).toContain('got this far')
  1129. })
  1130. it('CodeRunFailedError is a HarnessError with the CODE_RUN_FAILED code', () => {
  1131. const error = new CodeRunFailedError('boom')
  1132. expect(error.code).toBe('CODE_RUN_FAILED')
  1133. expect(error.name).toBe('CodeRunFailedError')
  1134. })
  1135. it('aborting the outer signal aborts the in-flight sub-dispatch and abandons queued ones', async () => {
  1136. const { ctx, runtime } = await setup({ mode: 'code' })
  1137. const seen: string[] = []
  1138. let sawAbort = false
  1139. ctx.tools.register(defineContentToolFixture({
  1140. name: 'slow',
  1141. description: 'Slow tool observing its signal.',
  1142. parameters: { id: { type: 'string', required: true } },
  1143. async execute(args, exec) {
  1144. seen.push(args.id)
  1145. await new Promise<void>((resolve) => {
  1146. const timer = setTimeout(resolve, 500)
  1147. exec.signal.addEventListener('abort', () => { sawAbort = true; clearTimeout(timer); resolve() }, { once: true })
  1148. })
  1149. return [{ type: 'text' as const, text: args.id }]
  1150. },
  1151. }))
  1152. const controller = new AbortController()
  1153. runtime.behavior = async (request) => {
  1154. const tools = request.bindings[0]!.functions
  1155. const calls = [tools.slow!({ id: 'first' }).catch(() => 'rejected'), tools.slow!({ id: 'second' }).catch(() => 'rejected')]
  1156. setTimeout(() => { controller.abort('user-cancel') }, 50)
  1157. await Promise.all(calls)
  1158. // A real runtime would be terminated by the abort; the fake honors the
  1159. // contract by reporting the abort as the run failure.
  1160. return { logs: [], error: { kind: 'abort', message: 'user-cancel' } }
  1161. }
  1162. const result = await runCode(ctx, 'program', { signal: controller.signal })
  1163. expect(result.isError).toBe(true)
  1164. expect((result.content[0] as { text: string }).text).toContain('code run failed (abort)')
  1165. expect(seen).toEqual(['first'])
  1166. expect(sawAbort).toBe(true)
  1167. })
  1168. it('a runtime that starts a binding call and then REJECTS still reaches quiescence before returning', async () => {
  1169. const { ctx, runtime } = await setup({ mode: 'code' })
  1170. const { agent, events } = fakeAgent()
  1171. let sawAbort = false
  1172. let started!: () => void
  1173. const inFlight = new Promise<void>((resolve) => { started = resolve })
  1174. ctx.tools.register(defineContentToolFixture({
  1175. name: 'slow',
  1176. description: 'Slow tool observing its signal.',
  1177. parameters: { id: { type: 'string', required: true } },
  1178. async execute(args, exec) {
  1179. started()
  1180. await new Promise<void>((resolve) => {
  1181. const timer = setTimeout(resolve, 500)
  1182. exec.signal.addEventListener('abort', () => { sawAbort = true; clearTimeout(timer); resolve() }, { once: true })
  1183. })
  1184. return [{ type: 'text' as const, text: args.id }]
  1185. },
  1186. }))
  1187. runtime.behavior = async (request) => {
  1188. // Start a sub-dispatch, keep its rejection held, and fail the run once the tool is
  1189. // genuinely in flight — a seam error after work has begun.
  1190. request.bindings[0]!.functions.slow!({ id: 'orphan' }).catch(() => 'held')
  1191. await inFlight
  1192. throw new Error('backend exploded')
  1193. }
  1194. const result = await runCode(ctx, 'program', { agent })
  1195. expect(result.isError).toBe(true)
  1196. expect((result.content[0] as { text: string }).text).toContain('backend exploded')
  1197. // Quiescence held: the in-flight sub-dispatch was aborted and its event
  1198. // logged INSIDE the run_code execution, not after it returned.
  1199. expect(sawAbort).toBe(true)
  1200. expect(events.filter(event => event.type === 'tool/code-dispatch').map(event => (event.data as { name: string }).name)).toEqual(['slow'])
  1201. })
  1202. it('runs without an owning agent: dispatches work, event logging is skipped', async () => {
  1203. const { ctx, runtime } = await setup({ mode: 'code' })
  1204. const calls = registerEcho(ctx)
  1205. runtime.behavior = async (request) => {
  1206. await request.bindings[0]!.functions.echo!({ value: 'x' })
  1207. return { logs: [], value: 'ok' }
  1208. }
  1209. const result = await runCode(ctx, 'program')
  1210. expect(result.isError).toBe(false)
  1211. expect(calls).toEqual([{ value: 'x' }])
  1212. })
  1213. it('executing run_code under a missing runtime is a structured isError, not a crash', async () => {
  1214. const ctx = new Context()
  1215. await ctx.plugin(SystemPrompt, {})
  1216. await ctx.plugin(ToolRuntime, { mode: 'code' })
  1217. const result = await runCode(ctx, 'program')
  1218. expect(result.isError).toBe(true)
  1219. expect((result.content[0] as { text: string }).text).toContain('requires a code runtime')
  1220. })
  1221. it('presents the model-authored description as the execute-card title over the program input', async () => {
  1222. const { ctx } = await setup({ mode: 'code' })
  1223. const tool = ctx.tools.get(RUN_CODE_NAME)!
  1224. // The description labels the card (the bash description precedent); the
  1225. // program itself remains the expanded raw input.
  1226. expect(tool.presentCall?.({ code: 'return 1', description: 'Return the constant one' })).toEqual({
  1227. card: 'generic',
  1228. title: 'Return the constant one',
  1229. kind: 'execute',
  1230. rawInput: 'return 1',
  1231. })
  1232. })
  1233. it('rejects a whitespace-only description with a structured isError', async () => {
  1234. const { ctx } = await setup({ mode: 'code' })
  1235. const result = await runCode(ctx, 'return 1', { description: ' ' })
  1236. expect(result.isError).toBe(true)
  1237. expect((result.content[0] as { text: string }).text).toContain('invalid description')
  1238. })
  1239. it.each([
  1240. ['logs only', { logs: ['printed'] }, 'printed'],
  1241. ['result only', { logs: [], value: 'returned' }, 'returned'],
  1242. ['logs plus result', { logs: ['printed'], value: 'returned' }, 'printed\nreturned'],
  1243. ['no output', { logs: [] }, '(run_code completed with no output)'],
  1244. ] as [string, CodeRunResult, string][])('keeps %s in durable content without a result presenter', async (_name, output, text) => {
  1245. const { ctx, runtime } = await setup({ mode: 'code' })
  1246. runtime.behavior = () => Promise.resolve(output)
  1247. const result = await runCode(ctx, 'return 1')
  1248. const tool = ctx.tools.get(RUN_CODE_NAME)!
  1249. expect(result.content).toEqual([{ type: 'text', text }])
  1250. // Presenters keep the pending program title and render this durable content
  1251. // through their generic fallback. Omitting a result view also prevents the
  1252. // host frame from carrying the same raw content a second time.
  1253. expect('presentResult' in tool).toBe(false)
  1254. })
  1255. it('keeps a post-policy spill preview in durable content without a result presenter', async () => {
  1256. const { ctx, runtime } = await setup({ mode: 'code' })
  1257. const preview = 'HEAD\n\n(Omitted 100 bytes. Full formatted result stored at: /tmp/run-code.txt.)\n\nTAIL'
  1258. runtime.behavior = () => Promise.resolve({ logs: ['printed'], value: 'returned' })
  1259. ctx.on('tools/post-execute', (exec, _result, next): Promise<PostToolDecision> => {
  1260. if (exec.name !== RUN_CODE_NAME) return next()
  1261. return Promise.resolve({ kind: 'accept', content: [{ type: 'text', text: preview }] })
  1262. })
  1263. const result = await runCode(ctx, 'return 1')
  1264. const tool = ctx.tools.get(RUN_CODE_NAME)!
  1265. expect(result.content).toEqual([{ type: 'text', text: preview }])
  1266. expect('presentResult' in tool).toBe(false)
  1267. })
  1268. it('keeps canonical failure content durable without a result presenter', async () => {
  1269. const { ctx, runtime } = await setup({ mode: 'code' })
  1270. runtime.behavior = () => Promise.resolve({
  1271. logs: ['captured before failure'],
  1272. error: { kind: 'output-limit', message: 'outer output exceeded 8 bytes' },
  1273. })
  1274. const result = await runCode(ctx, 'return 1')
  1275. const tool = ctx.tools.get(RUN_CODE_NAME)!
  1276. expect(result.isError).toBe(true)
  1277. expect(result.content).toEqual([{
  1278. type: 'text',
  1279. text: 'Error: code run failed (output-limit): outer output exceeded 8 bytes\nCaptured output:\ncaptured before failure',
  1280. }])
  1281. expect('presentResult' in tool).toBe(false)
  1282. })
  1283. it('logs the complete sub-result content verbatim, non-text blocks and long text included', async () => {
  1284. const { ctx, runtime } = await setup({ mode: 'code' })
  1285. const { agent, events } = fakeAgent()
  1286. const long = 'x'.repeat(300)
  1287. ctx.tools.register(defineTool({
  1288. name: 'mixed',
  1289. description: 'Returns mixed content.',
  1290. parameters: {},
  1291. output: {
  1292. schema: { type: 'string' },
  1293. render: () => [
  1294. { type: 'text', text: long },
  1295. { type: 'reasoning', text: 'hidden' },
  1296. ],
  1297. },
  1298. execute() {
  1299. return Promise.resolve('mixed-value')
  1300. },
  1301. }))
  1302. runtime.behavior = async (request) => {
  1303. const value = await request.bindings[0]!.functions.mixed!({})
  1304. return { logs: [], value }
  1305. }
  1306. const result = await runCode(ctx, 'program', { agent })
  1307. expect(result.isError).toBe(false)
  1308. expect((result.content[0] as { text: string }).text).toBe('mixed-value')
  1309. const dispatch = events.find(event => event.type === 'tool/code-dispatch')?.data as SessionEventMap['tool/code-dispatch']
  1310. expect(dispatch.content).toEqual([
  1311. { type: 'text', text: long },
  1312. { type: 'reasoning', text: 'hidden' },
  1313. ])
  1314. })
  1315. it('rejects undefined, getter-throwing, exotic, and unrepresentable binding arguments before dispatch', async () => {
  1316. const { ctx, runtime } = await setup({ mode: 'code' })
  1317. const calls = registerEcho(ctx)
  1318. const { agent, events } = fakeAgent()
  1319. runtime.behavior = async (request) => {
  1320. const echo = request.bindings[0]!.functions.echo!
  1321. const catchMessage = (promise: Promise<unknown>) => promise.then(() => 'resolved', (error: unknown) => error instanceof Error ? error.message : String(error))
  1322. return {
  1323. logs: [],
  1324. value: [
  1325. // Root undefined must reject up front: the event log rejects it as
  1326. // data, and nothing may execute unlogged.
  1327. await catchMessage(echo(undefined)),
  1328. await catchMessage(echo(Object.defineProperty({}, 'bad', { enumerable: true, get() { throw 'raw-throw' } }))),
  1329. await catchMessage(echo(Object.defineProperty({}, 'bad', { enumerable: true, get() { throw new Error('error-throw') } }))),
  1330. await catchMessage(echo(new Date(0))),
  1331. // A bare function is a value JSON cannot represent at all.
  1332. await catchMessage(echo(() => 1)),
  1333. ].join(' | '),
  1334. }
  1335. }
  1336. const result = await runCode(ctx, 'program', { agent })
  1337. const text = (result.content[0] as { text: string }).text
  1338. expect(text).toContain('call the tool with an arguments object')
  1339. expect(text).toContain('lossless JSON: raw-throw')
  1340. expect(text).toContain('lossless JSON: error-throw')
  1341. expect(text.match(/tool arguments must be lossless JSON/g)).toHaveLength(5)
  1342. // None dispatched or logged.
  1343. expect(calls).toEqual([])
  1344. expect(events.filter(event => event.type === 'tool/code-dispatch')).toEqual([])
  1345. })
  1346. it('dispatches and durably logs binding arguments deeper than the structured-clone call stack', async () => {
  1347. const { ctx, runtime } = await setup({ mode: 'code' })
  1348. const depth = 5_000
  1349. let observedDepth = 0
  1350. let observedLeaf: JsonValue | undefined
  1351. ctx.tools.register(defineTool({
  1352. name: 'deep_args',
  1353. description: 'Measure a deeply nested JSON argument.',
  1354. parameters: { nested: { type: 'json', required: true } },
  1355. output: {
  1356. schema: { type: 'integer' },
  1357. render: (_args, value) => [{ type: 'text', text: String(value) }],
  1358. },
  1359. execute(args) {
  1360. let cursor = args.nested
  1361. while (Array.isArray(cursor)) {
  1362. if (cursor.length !== 1) throw new Error('expected one item per nesting layer')
  1363. observedDepth++
  1364. cursor = cursor[0]!
  1365. }
  1366. observedLeaf = cursor
  1367. return Promise.resolve(observedDepth)
  1368. },
  1369. }))
  1370. const session = Session.create(SessionId('deep-code-arguments'))
  1371. const agent = { session } as Agent
  1372. runtime.behavior = async (request) => {
  1373. let nested: JsonValue = 'leaf'
  1374. for (let index = 0; index < depth; index++) nested = [nested]
  1375. const value = await request.bindings[0]!.functions.deep_args!({ nested })
  1376. return { logs: [], value }
  1377. }
  1378. const result = await runCode(ctx, 'return tools.deep_args(...)', { agent })
  1379. expect(result.isError).toBe(false)
  1380. expect(result.isError ? undefined : result.value).toEqual({ logs: [], result: depth })
  1381. expect({ observedDepth, observedLeaf }).toEqual({ observedDepth: depth, observedLeaf: 'leaf' })
  1382. const dispatch = session.events.find(event => event.type === 'tool/code-dispatch')
  1383. if (dispatch === undefined) throw new Error('expected a durable tool/code-dispatch event')
  1384. const logged = dispatch.data.arguments as { nested: JsonValue }
  1385. let loggedDepth = 0
  1386. let loggedCursor = logged.nested
  1387. while (Array.isArray(loggedCursor)) {
  1388. if (loggedCursor.length !== 1) throw new Error('expected one logged item per nesting layer')
  1389. loggedDepth++
  1390. loggedCursor = loggedCursor[0]!
  1391. }
  1392. expect({ loggedDepth, loggedCursor }).toEqual({ loggedDepth: depth, loggedCursor: 'leaf' })
  1393. })
  1394. it('gives the tool and durable log the same immutable argument value', async () => {
  1395. const { ctx, runtime } = await setup({ mode: 'code' })
  1396. const { agent, events } = fakeAgent()
  1397. let mutationSucceeded: boolean | undefined
  1398. ctx.tools.register(defineContentToolFixture({
  1399. name: 'mutator',
  1400. description: 'Attempts to mutate its args object.',
  1401. parameters: { list: { type: 'array', required: true } },
  1402. execute(args) {
  1403. mutationSucceeded = Reflect.set(args.list, 1, 'injected-by-tool')
  1404. return Promise.resolve([{ type: 'text' as const, text: 'protected' }])
  1405. },
  1406. }))
  1407. runtime.behavior = async (request) => {
  1408. await request.bindings[0]!.functions.mutator!({ list: ['original'] })
  1409. return { logs: [] }
  1410. }
  1411. const result = await runCode(ctx, 'program', { agent })
  1412. expect(result.isError).toBe(false)
  1413. expect(mutationSucceeded).toBe(false)
  1414. const dispatch = events.find(event => event.type === 'tool/code-dispatch')?.data as SessionEventMap['tool/code-dispatch']
  1415. expect(dispatch.arguments).toEqual({ list: ['original'] })
  1416. })
  1417. it('exposes a tool named __proto__ as an ordinary own binding', async () => {
  1418. const { ctx, runtime } = await setup({ mode: 'code' })
  1419. ctx.tools.register(defineTool({
  1420. name: '__proto__',
  1421. description: 'A prototype-colliding tool name.',
  1422. parameters: {},
  1423. output: {
  1424. schema: { type: 'string' },
  1425. render: (_args, value) => [{ type: 'text', text: value }],
  1426. },
  1427. execute() { return Promise.resolve('proto-tool-ok') },
  1428. }))
  1429. runtime.behavior = async (request) => {
  1430. const functions = request.bindings[0]!.functions
  1431. expect(Object.getPrototypeOf(functions)).toBeNull()
  1432. const value = await functions['__proto__']!({})
  1433. return { logs: [], value }
  1434. }
  1435. const result = await runCode(ctx, 'program')
  1436. expect(result.isError).toBe(false)
  1437. expect(result.content[0]).toEqual({ type: 'text', text: 'proto-tool-ok' })
  1438. })
  1439. it('renders every non-string JSON root as pretty JSON while preserving strings raw', async () => {
  1440. const { ctx, runtime } = await setup({ mode: 'code' })
  1441. runtime.behavior = () => Promise.resolve({ logs: [], value: { n: 42, ok: true } })
  1442. expect((await runCode(ctx, 'object')).content[0]).toEqual({ type: 'text', text: '{\n "n": 42,\n "ok": true\n}' })
  1443. runtime.behavior = () => Promise.resolve({ logs: [], value: {} })
  1444. expect((await runCode(ctx, 'empty object')).content[0]).toEqual({ type: 'text', text: '{}' })
  1445. const nested = { outer: [{ inner: true }] }
  1446. runtime.behavior = () => Promise.resolve({ logs: [], value: nested })
  1447. expect((await runCode(ctx, 'nested')).content[0]).toEqual({ type: 'text', text: JSON.stringify(nested, null, 2) })
  1448. runtime.behavior = () => Promise.resolve({ logs: [], value: ['x', 2] })
  1449. expect((await runCode(ctx, 'array')).content[0]).toEqual({ type: 'text', text: '[\n "x",\n 2\n]' })
  1450. runtime.behavior = () => Promise.resolve({ logs: [], value: [] })
  1451. expect((await runCode(ctx, 'empty array')).content[0]).toEqual({ type: 'text', text: '[]' })
  1452. runtime.behavior = () => Promise.resolve({ logs: [], value: null })
  1453. expect((await runCode(ctx, 'null')).content[0]).toEqual({ type: 'text', text: 'null' })
  1454. runtime.behavior = () => Promise.resolve({ logs: [], value: 'raw' })
  1455. expect((await runCode(ctx, 'string')).content[0]).toEqual({ type: 'text', text: 'raw' })
  1456. runtime.behavior = () => Promise.resolve({ logs: [] })
  1457. const absent = await runCode(ctx, 'undefined')
  1458. expect(absent.content[0]).toEqual({ type: 'text', text: '(run_code completed with no output)' })
  1459. expect(absent.isError ? undefined : absent.value).toEqual({ logs: [] })
  1460. })
  1461. it('renders deeply nested JSON without recursive traversal or quadratic indentation', async () => {
  1462. const { ctx, runtime } = await setup({ mode: 'code' })
  1463. let value: JsonValue = {
  1464. emptyArray: [],
  1465. emptyObject: {},
  1466. pair: ['leaf', 2],
  1467. record: { first: true, second: null },
  1468. }
  1469. for (let depth = 0; depth < 5_000; depth++) value = [value]
  1470. runtime.behavior = () => Promise.resolve({ logs: [], value })
  1471. const result = await runCode(ctx, 'deep result')
  1472. expect(result.isError).toBe(false)
  1473. const text = (result.content[0] as { type: 'text'; text: string }).text
  1474. expect(text.startsWith('[\n [\n [')).toBe(true)
  1475. expect(text).toContain('"leaf"')
  1476. expect(text.endsWith(']')).toBe(true)
  1477. expect(text.length).toBeLessThan(11_000)
  1478. })
  1479. it('short-circuits a pre-aborted outer signal before the code runtime', async () => {
  1480. const { ctx, runtime } = await setup({ mode: 'code' })
  1481. const calls = registerEcho(ctx)
  1482. runtime.behavior = (request) => {
  1483. // The fake honors the seam contract for an already-aborted signal.
  1484. if (request.signal?.aborted) return Promise.resolve({ logs: [], error: { kind: 'abort' as const, message: String(request.signal.reason) } })
  1485. return Promise.resolve({ logs: [], value: 'unreachable' })
  1486. }
  1487. const controller = new AbortController()
  1488. controller.abort('too-late')
  1489. const result = await runCode(ctx, 'program', { signal: controller.signal })
  1490. expect(result.isError).toBe(true)
  1491. expect(result).toEqual({
  1492. content: [{ type: 'text', text: 'Error: tool call aborted before dispatch' }],
  1493. isError: true,
  1494. error: {
  1495. message: 'tool call aborted before dispatch',
  1496. info: { name: 'AbortError', code: TOOL_ABORTED_BEFORE_DISPATCH },
  1497. },
  1498. })
  1499. expect(runtime.lastRequest).toBeUndefined()
  1500. expect(calls).toEqual([])
  1501. })
  1502. it('reports cancellation after rejecting a late binding without dispatching it', async () => {
  1503. const { ctx, runtime } = await setup({ mode: 'code' })
  1504. const calls = registerEcho(ctx)
  1505. const controller = new AbortController()
  1506. runtime.behavior = async (request) => {
  1507. controller.abort('cancelled-mid-run')
  1508. const message = await request.bindings[0]!.functions.echo!({ value: 'x' })
  1509. .then(() => 'resolved', (error: unknown) => error instanceof Error ? error.message : String(error))
  1510. return { logs: [], value: message }
  1511. }
  1512. const result = await runCode(ctx, 'program', { signal: controller.signal })
  1513. expect(result.isError).toBe(true)
  1514. expect(result.error).toEqual({
  1515. message: 'tool call aborted',
  1516. info: { name: 'AbortError', code: 'ABORTED' },
  1517. })
  1518. expect((result.content[0] as { text: string }).text).toBe('Error: tool call aborted')
  1519. expect(calls).toEqual([])
  1520. })
  1521. it('a tool/code-dispatch event never derives a model message', () => {
  1522. const session = Session.create(SessionId('code-mode-derive'))
  1523. session.append('user/message', createUserMessage({
  1524. content: [{ type: 'text', text: 'hi' }], source: { kind: 'user' },
  1525. }), { surfaceOp: 'append' })
  1526. session.append('tool/code-dispatch', {
  1527. rootCallId: ToolCallId('p1'),
  1528. parentCallId: ToolCallId('p1'),
  1529. subCallId: ToolCallId('p1:code:1'),
  1530. name: 'echo',
  1531. arguments: { value: 'x' },
  1532. isError: false,
  1533. content: [{ type: 'text', text: 'echo:x' }],
  1534. })
  1535. const derived = session.deriveMessages()
  1536. expect(derived).toHaveLength(1)
  1537. expect(derived[0]?.role).toBe('user')
  1538. })
  1539. it('direct construction rejects a non-positive parallel sub-call cap at load', async () => {
  1540. const ctx = new Context()
  1541. await ctx.plugin(SystemPrompt, {})
  1542. expect(() => new ToolRuntime(ctx, { mode: 'code', maxParallelSubCalls: 0 }))
  1543. .toThrow('maxParallelSubCalls must be a positive integer')
  1544. })
  1545. it('direct construction in code mode defaults the parallel sub-call cap', async () => {
  1546. const ctx = new Context()
  1547. await ctx.plugin(SystemPrompt, {})
  1548. const registry = new ToolRuntime(ctx, { mode: 'code' })
  1549. expect(registry.get(RUN_CODE_NAME)).toBeDefined()
  1550. })
  1551. it('defaults to native mode under direct construction with no config', async () => {
  1552. const ctx = new Context()
  1553. await ctx.plugin(SystemPrompt, {})
  1554. const registry = new ToolRuntime(ctx)
  1555. expect(registry.get(RUN_CODE_NAME)).toBeUndefined()
  1556. const assembly = await ctx.systemPrompt.assemble()
  1557. expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
  1558. })
  1559. it('denies a model-direct native-tool call under code mode as UNKNOWN_TOOL', async () => {
  1560. const ctx = new Context()
  1561. await ctx.plugin(SystemPrompt, {})
  1562. const registry = new ToolRuntime(ctx, { mode: 'code' })
  1563. registerEcho(ctx, 'write')
  1564. const result = await registry.execute({
  1565. signal: testToolSignal,
  1566. callId: ToolCallId('call-1'),
  1567. name: 'write',
  1568. arguments: { text: 'hello' },
  1569. })
  1570. expect(result.isError).toBe(true)
  1571. expect(result.error?.info).toEqual({ name: 'ToolNotFoundError', code: 'UNKNOWN_TOOL' })
  1572. // The name IS declared to this model, so a bare `unknown tool` reads as a
  1573. // broken deployment. The denial carries the route instead.
  1574. expect(result.error?.message).toBe(
  1575. `unknown tool "write": only \`${RUN_CODE_NAME}\` is callable directly — call \`write\` from inside a \`${RUN_CODE_NAME}\` program instead`,
  1576. )
  1577. })
  1578. it('routes a pre-aborted collapsed call through ABORTED_BEFORE_DISPATCH', async () => {
  1579. const ctx = new Context()
  1580. await ctx.plugin(SystemPrompt, {})
  1581. const registry = new ToolRuntime(ctx, { mode: 'code' })
  1582. registerEcho(ctx, 'write')
  1583. const aborted = new AbortController()
  1584. aborted.abort()
  1585. const result = await registry.execute({
  1586. signal: aborted.signal,
  1587. callId: ToolCallId('call-1'),
  1588. name: 'write',
  1589. arguments: { text: 'hello' },
  1590. })
  1591. expect(result.isError).toBe(true)
  1592. expect(result.error?.info?.code).toBe(TOOL_ABORTED_BEFORE_DISPATCH)
  1593. })
  1594. })
  1595. /**
  1596. * Presentation is per agent, because an agent preset composes it: one
  1597. * deployment runs a Code Mode agent beside native ones, and neither may see
  1598. * the other's catalog. The deployment `mode` is the default those agents
  1599. * shadow, not a process-wide fact.
  1600. */
  1601. describe('per-agent presentation', () => {
  1602. it('gives one agent Code Mode while the deployment stays native', async () => {
  1603. const { ctx, systemPrompt } = await setup({ mode: 'native' })
  1604. const calls = registerEcho(ctx)
  1605. const { scope, agent } = await mintAgentScope(ctx)
  1606. scope.ctx.tools.presentAs('code')
  1607. const coded = await systemPrompt.assemble({ scope: agent })
  1608. expect(coded.tools.map(tool => tool.name)).toEqual([RUN_CODE_NAME])
  1609. expect(coded.sections.find(section => section.name === 'tools:sdk')?.text)
  1610. .toContain('echo')
  1611. // Announced surface and callable surface must agree for THIS agent, whose
  1612. // mode is its own rather than the deployment's.
  1613. const denied = await ctx.tools.execute({
  1614. signal: testToolSignal,
  1615. callId: ToolCallId('coded-direct'),
  1616. name: 'echo',
  1617. arguments: { value: 'coded' },
  1618. agent,
  1619. })
  1620. expect(denied.error?.info).toEqual({ name: 'ToolNotFoundError', code: 'UNKNOWN_TOOL' })
  1621. expect(calls).toEqual([])
  1622. // The deployment default is untouched: an agent that declared nothing —
  1623. // and the global view behind it — still sees the native catalog.
  1624. const native = await systemPrompt.assemble()
  1625. expect(native.tools.map(tool => tool.name)).toEqual(['echo'])
  1626. expect(native.sections.some(section => section.name === 'tools:sdk')).toBe(false)
  1627. })
  1628. it('inherits a STANDING preset scope\'s mode down the chain, agents beside it unaffected', async () => {
  1629. const { bindScopeParent } = await import('@deepseek-ai/dsh-scope')
  1630. const { ctx, systemPrompt } = await setup({ mode: 'native' })
  1631. const calls = registerEcho(ctx)
  1632. // The preset's standing scope declares once; the agent only PARENTS to it
  1633. // (the per-preset standing mount configuration has no per-agent declaration).
  1634. const standing = await mintAgentScope(ctx, 'preset:code-like')
  1635. standing.scope.ctx.tools.presentAs('code')
  1636. const joined = await mintAgentScope(ctx, 'joined-agent')
  1637. bindScopeParent(joined.agent, standing.agent)
  1638. const loner = await mintAgentScope(ctx, 'loner-agent')
  1639. expect(ctx.tools.get(RUN_CODE_NAME, joined.agent)).toBeDefined()
  1640. const coded = await systemPrompt.assemble({ scope: joined.agent })
  1641. expect(coded.tools.map(tool => tool.name)).toEqual([RUN_CODE_NAME])
  1642. // Through the EXECUTOR, not just the wire: the deployment default is
  1643. // `native` here, so a collapse predicate reading it instead of this
  1644. // scope's effective mode would announce [run_code] and still execute the
  1645. // native call — the bypass, reopened for exactly the preset composition
  1646. // `dsh-agent-tool-presentation` produces.
  1647. expect(ctx.tools.executionMode({
  1648. signal: testToolSignal,
  1649. callId: ToolCallId('preset-coded-schedule'),
  1650. name: 'echo',
  1651. arguments: { value: 'joined' },
  1652. agent: joined.agent,
  1653. })).toEqual({ kind: 'exclusive' })
  1654. const denied = await ctx.tools.execute({
  1655. signal: testToolSignal,
  1656. callId: ToolCallId('preset-coded-direct'),
  1657. name: 'echo',
  1658. arguments: { value: 'joined' },
  1659. agent: joined.agent,
  1660. })
  1661. expect(denied.error?.info).toEqual({ name: 'ToolNotFoundError', code: 'UNKNOWN_TOOL' })
  1662. expect(calls).toEqual([])
  1663. // A sibling that never parented stays native, as does the global view.
  1664. expect(ctx.tools.get(RUN_CODE_NAME, loner.agent)).toBeUndefined()
  1665. const native = await systemPrompt.assemble({ scope: loner.agent })
  1666. expect(native.tools.map(tool => tool.name)).toEqual(['echo'])
  1667. const allowed = await ctx.tools.execute({
  1668. signal: testToolSignal,
  1669. callId: ToolCallId('native-sibling-direct'),
  1670. name: 'echo',
  1671. arguments: { value: 'loner' },
  1672. agent: loner.agent,
  1673. })
  1674. expect(allowed).toMatchObject({ isError: false, value: 'echo:loner' })
  1675. expect(calls).toEqual([{ value: 'loner' }])
  1676. })
  1677. it('keeps run_code out of a native agent\'s dispatch table', async () => {
  1678. const { ctx } = await setup({ mode: 'native' })
  1679. registerEcho(ctx)
  1680. const coded = await mintAgentScope(ctx, 'coded')
  1681. const plain = await mintAgentScope(ctx, 'plain')
  1682. coded.scope.ctx.tools.presentAs('code')
  1683. // Not merely hidden from the prompt: the transport one agent presents must
  1684. // not be dispatchable by another that never presented it.
  1685. expect(ctx.tools.get(RUN_CODE_NAME, coded.agent)).toBeDefined()
  1686. expect(ctx.tools.get(RUN_CODE_NAME, plain.agent)).toBeUndefined()
  1687. expect(ctx.tools.get(RUN_CODE_NAME)).toBeUndefined()
  1688. })
  1689. it('lets an agent opt out of a code-mode deployment', async () => {
  1690. const { ctx, systemPrompt } = await setup({ mode: 'code' })
  1691. registerEcho(ctx)
  1692. const { scope, agent } = await mintAgentScope(ctx)
  1693. scope.ctx.tools.presentAs('native')
  1694. const assembly = await systemPrompt.assemble({ scope: agent })
  1695. expect(assembly.tools.map(tool => tool.name)).toEqual(['echo'])
  1696. // The deployment's global section still reaches this scope; rendering it
  1697. // empty is what keeps the opted-out agent's prompt free of an SDK.
  1698. expect(assembly.sections.find(section => section.name === 'tools:sdk')?.text).toBe('')
  1699. })
  1700. it('restores the deployment default when the agent unloads', async () => {
  1701. const { ctx, systemPrompt } = await setup({ mode: 'native' })
  1702. registerEcho(ctx)
  1703. const { scope, agent } = await mintAgentScope(ctx)
  1704. const dispose = scope.ctx.tools.presentAs('code')
  1705. dispose()
  1706. const assembly = await systemPrompt.assemble({ scope: agent })
  1707. expect(assembly.tools.map(tool => tool.name)).toEqual(['echo'])
  1708. expect(assembly.sections.some(section => section.name === 'tools:sdk')).toBe(false)
  1709. })
  1710. it('refuses a second declaration for the same agent', async () => {
  1711. const { ctx } = await setup({ mode: 'native' })
  1712. const { scope } = await mintAgentScope(ctx)
  1713. scope.ctx.tools.presentAs('code')
  1714. // Two answers to "which form does the model see" is a contradiction, and
  1715. // silently keeping either one would make the composition unreadable.
  1716. expect(() => scope.ctx.tools.presentAs('both'))
  1717. .toThrow('conflicts with "code" already declared')
  1718. })
  1719. it('refuses an unscoped declaration', async () => {
  1720. const { ctx } = await setup({ mode: 'native' })
  1721. expect(() => ctx.tools.presentAs('code'))
  1722. .toThrow('requires a scoped context')
  1723. })
  1724. it('reserves run_code even where no agent presents it', async () => {
  1725. const { ctx } = await setup({ mode: 'native' })
  1726. // The name must stay free under a native deployment too: an agent preset
  1727. // mounting later would otherwise collide with whatever took it.
  1728. expect(() => registerEcho(ctx, RUN_CODE_NAME)).toThrow('is reserved')
  1729. })
  1730. it('reports the missing runtime against the agent\'s own mode', async () => {
  1731. const { ctx, systemPrompt } = await setup({ mode: 'native', runtime: false })
  1732. registerEcho(ctx)
  1733. const { scope, agent } = await mintAgentScope(ctx)
  1734. scope.ctx.tools.presentAs('both')
  1735. await expect(systemPrompt.assemble({ scope: agent }))
  1736. .rejects.toThrow('mode "both" requires a code runtime')
  1737. })
  1738. })