structured.spec.ts 36 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm'
  4. import SessionStore from '@deepseek-ai/dsh-session'
  5. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  6. import ToolRegistry from '@deepseek-ai/dsh-tools'
  7. import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
  8. import type { ContinuationDecision } from '@deepseek-ai/dsh-agent'
  9. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  10. import * as Invariants from '@deepseek-ai/dsh-invariants'
  11. import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent'
  12. import type { Config as ToolConfig, StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
  13. import { RUN_CODE_NAME } from '@deepseek-ai/dsh-tools'
  14. import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
  15. import { startInProcessRun } from '../src/index.ts'
  16. import {
  17. STRUCTURED_OUTPUT_INSTRUCTION,
  18. STRUCTURED_OUTPUT_TOOL,
  19. } from '../src/structured.ts'
  20. type Script = ConstructorParameters<typeof MockAdapter>[0]
  21. interface CodeRunRequestLike {
  22. bindings: { global: string; functions: Record<string, (args: unknown) => Promise<unknown>> }[]
  23. }
  24. interface SetupOptions {
  25. toolMode?: ToolConfig['mode']
  26. codeRun?: (request: CodeRunRequestLike) => Promise<{ logs: never[]; value?: unknown }>
  27. }
  28. const SCHEMA: StructuredOutputSchema = {
  29. type: 'object',
  30. properties: { answer: { type: 'number' }, note: { type: 'string' } },
  31. required: ['answer'],
  32. }
  33. /**
  34. * Real loop + scripted mock model + an INLINE fresh-conversation provider over the
  35. * shared driver. The concrete backend plugins are deliberately NOT loaded —
  36. * they would devDep-cycle this package (spawn/fork already depend on the
  37. * driver), and the runtime under test is the driver's; plugin-level structured
  38. * coverage lives in the spawn/fork specs. The mock model script drives the
  39. * child's structured_output calls.
  40. */
  41. async function setup(script: Script, options: SetupOptions = {}) {
  42. const ctx = new Context()
  43. const adapter = new MockAdapter(script)
  44. await ctx.plugin(LlmService)
  45. await ctx.plugin(SessionStore)
  46. await ctx.plugin(SystemPrompt)
  47. await ctx.plugin(ToolRegistry, { mode: options.toolMode ?? 'native' })
  48. if (options.toolMode === 'code' || options.toolMode === 'both') {
  49. ctx.provide('codeRuntime', {
  50. language: 'typescript',
  51. isolation: 'test',
  52. run: options.codeRun ?? (() => Promise.resolve({ logs: [] })),
  53. } as never)
  54. }
  55. await ctx.plugin(AgentRegistry)
  56. await ctx.plugin(Invariants)
  57. await ctx.plugin(AgentLoop, { agents: [] })
  58. await ctx.plugin(SubagentService)
  59. const disposeProvider = ctx.subagents.registerProvider({
  60. name: 'spawn',
  61. capabilities: { outputSchema: true, depthLimit: true, toolFilter: false, persona: false },
  62. inheritsParentContext: false,
  63. start: (request: SubagentStartRequest) => startInProcessRun(request, {}),
  64. })
  65. ctx.llm.registerAdapter(['mock'], adapter)
  66. const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' })
  67. return { ctx, parent, adapter, disposeProvider }
  68. }
  69. function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial<SubagentStartRequest>): SubagentStartRequest {
  70. return {
  71. prompt: [{ type: 'text', text: 'produce the answer' }],
  72. parent,
  73. signal: new AbortController().signal,
  74. outputSchema: SCHEMA,
  75. ...extra,
  76. }
  77. }
  78. /** The tool names of one recorded model request. */
  79. function toolNames(request: GenerateOptions): string[] {
  80. return (request.tools ?? []).map(tool => tool.name)
  81. }
  82. describe('in-process structured output', () => {
  83. it('captures a valid structured_output call and surfaces result.structured', async () => {
  84. const { ctx, parent } = await setup([
  85. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }),
  86. ])
  87. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  88. const result = await run.result
  89. expect(result.stopReason).toBe('completed')
  90. expect(result.structured).toEqual({ answer: 42, note: 'done' })
  91. await run.dispose()
  92. })
  93. it('stops the turn after a successful capture — no extra model step is spent', async () => {
  94. const { ctx, parent, adapter } = await setup([
  95. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  96. textResponse('MUST NOT BE CONSUMED'),
  97. ])
  98. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  99. await run.result
  100. // Default continuation would run a second step after the tool call; the
  101. // structured runtime's turn-continuation veto stops the turn instead.
  102. expect(adapter.requests.length).toBe(1)
  103. await run.dispose()
  104. })
  105. it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => {
  106. // One model response carrying structured_output FIRST and a side-effecting
  107. // call after it: the continuation veto only fires at step end, so without
  108. // the pre-execute deny the trailing call would still run after the final
  109. // answer was accepted.
  110. const response = [
  111. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  112. { type: 'block-start', index: 1, blockType: 'tool-call' },
  113. { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
  114. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  115. { type: 'finish', reason: { kind: 'tool-calls' } },
  116. ] as Script[number]
  117. const { ctx, parent } = await setup([response])
  118. let sideEffectRan = false
  119. ctx.tools.register({
  120. name: 'side_effect',
  121. description: 'probe',
  122. parameters: { type: 'object', properties: {} },
  123. execute(): Promise<ContentBlock[]> {
  124. sideEffectRan = true
  125. return Promise.resolve([{ type: 'text', text: 'ran' }])
  126. },
  127. })
  128. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  129. const result = await run.result
  130. expect(result.stopReason).toBe('completed')
  131. expect(result.structured).toEqual({ answer: 5 })
  132. // The deny skipped dispatch entirely: the probe body never ran.
  133. expect(sideEffectRan).toBe(false)
  134. await run.dispose()
  135. })
  136. it('a later prepended pre-execute listener cannot resurrect dispatch after capture', async () => {
  137. const response = [
  138. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  139. { type: 'block-start', index: 1, blockType: 'tool-call' },
  140. { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
  141. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  142. { type: 'finish', reason: { kind: 'tool-calls' } },
  143. ] as Script[number]
  144. const { ctx, parent } = await setup([response])
  145. let sideEffectRan = false
  146. ctx.tools.register({
  147. name: 'side_effect',
  148. description: 'probe',
  149. parameters: { type: 'object', properties: {} },
  150. execute(): Promise<ContentBlock[]> {
  151. sideEffectRan = true
  152. return Promise.resolve([{ type: 'text', text: 'ran' }])
  153. },
  154. })
  155. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  156. // Registered after the child and prepended: this listener returns allow
  157. // after every downstream pre-execute decision. The service-owned guard
  158. // runs after the waterfall and can only deny, so the body still cannot run.
  159. ctx.on('tools/pre-execute', async (_exec, next) => {
  160. await next()
  161. return { kind: 'allow' as const }
  162. }, { prepend: true })
  163. const result = await run.result
  164. expect(result.structured).toEqual({ answer: 5 })
  165. expect(sideEffectRan).toBe(false)
  166. const child = ctx.agents.get(run.id)
  167. const sideEffectResult = child?.session.events.find(event =>
  168. event.type === 'tool/result' && event.data.callId === 'c2')
  169. expect(sideEffectResult?.type === 'tool/result' && sideEffectResult.data.isError).toBe(true)
  170. await run.dispose()
  171. })
  172. it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => {
  173. const response = [
  174. { type: 'block-start', index: 0, blockType: 'tool-call' },
  175. { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'side_effect', arguments: '{}' } },
  176. ...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk =>
  177. 'index' in chunk ? { ...chunk, index: 1 } : chunk),
  178. ] as Script[number]
  179. const { ctx, parent } = await setup([response])
  180. let sideEffectRan = false
  181. ctx.tools.register({
  182. name: 'side_effect',
  183. description: 'probe',
  184. parameters: { type: 'object', properties: {} },
  185. execute(): Promise<ContentBlock[]> {
  186. sideEffectRan = true
  187. return Promise.resolve([{ type: 'text', text: 'ran' }])
  188. },
  189. })
  190. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  191. const result = await run.result
  192. // The call ran BEFORE captured was set: the deny gate only guards the
  193. // window after the terminal answer landed.
  194. expect(sideEffectRan).toBe(true)
  195. expect(result.structured).toEqual({ answer: 6 })
  196. await run.dispose()
  197. })
  198. it('a later-prepended continuation wrapper cannot resurrect a captured turn', async () => {
  199. const { ctx, parent, adapter } = await setup([
  200. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  201. textResponse('MUST NOT BE CONSUMED'),
  202. ])
  203. ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'stop' }))
  204. let wrapperInstalled = false
  205. // Register before the ready-only start. The child session-start boundary is
  206. // after unpublished setup attached structured output but before the loop
  207. // can run. The wrapper awaits the
  208. // explicit downstream stop above, then overwrites that result with continue.
  209. // The later terminal checkpoint still wins.
  210. ctx.on('agent/session-start', (child) => {
  211. if (child === parent) return
  212. wrapperInstalled = true
  213. child.ctx.on('agent/turn-continuation', async (_subject, _turn, _decision, next): Promise<ContinuationDecision> => {
  214. const downstream = await next()
  215. expect(downstream).toEqual({ action: 'stop' })
  216. return { action: 'continue' }
  217. }, { prepend: true })
  218. })
  219. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  220. const result = await run.result
  221. expect(wrapperInstalled).toBe(true)
  222. expect(result.structured).toEqual({ answer: 7 })
  223. expect(result.stopReason).toBe('completed')
  224. expect(adapter.requests).toHaveLength(1)
  225. await run.dispose()
  226. })
  227. it('a continuation wrapper cannot carry steering past a captured terminal stop', async () => {
  228. const { ctx, parent, adapter } = await setup([
  229. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }),
  230. textResponse('MUST NOT BE CONSUMED'),
  231. ])
  232. // The downstream ordinary policy says stop. A wrapper registered after
  233. // start() delegates to that stop, then queues steering; ordinary folding
  234. // would turn the stop back into continue. The terminal checkpoint runs
  235. // afterwards and discards that steering.
  236. ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'stop' }))
  237. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  238. ctx.on('agent/session-start', (child) => {
  239. if (child.id !== run.id) return
  240. child.ctx.on('agent/turn-continuation', async (subject, _turn, _decision, next): Promise<ContinuationDecision> => {
  241. const downstream = await next()
  242. expect(downstream).toEqual({ action: 'stop' })
  243. subject.steer([{ type: 'text', text: 'late steering after downstream stop' }])
  244. return downstream
  245. }, { prepend: true })
  246. })
  247. const result = await run.result
  248. const child = ctx.agents.get(run.id)
  249. expect(result.structured).toEqual({ answer: 9 })
  250. expect(adapter.requests).toHaveLength(1)
  251. expect(child?.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1)
  252. expect(child?.session.events.filter(event => event.type === 'steering/message')).toHaveLength(0)
  253. await run.dispose()
  254. })
  255. it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => {
  256. const { ctx, parent } = await setup([
  257. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }),
  258. toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  259. ])
  260. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  261. const result = await run.result
  262. expect(result.structured).toEqual({ answer: 7 })
  263. expect(result.stopReason).toBe('completed')
  264. // The child's log carries the isError tool/result for the invalid call.
  265. const child = ctx.agents.get(run.id)!
  266. const results = child.session.events.filter(e => e.type === 'tool/result')
  267. expect(results.length).toBe(2)
  268. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  269. await run.dispose()
  270. })
  271. it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => {
  272. const { ctx, parent, adapter } = await setup([
  273. textResponse('here is my answer in prose'),
  274. textResponse('MUST NOT BE CONSUMED'),
  275. ])
  276. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  277. const result = await run.result
  278. expect(result.stopReason).toBe('error')
  279. expect(result.structured).toBeUndefined()
  280. // Exactly one model request and one user message: no nudge turn exists.
  281. expect(adapter.requests.length).toBe(1)
  282. const child = ctx.agents.get(run.id)!
  283. expect(child.session.events.filter(e => e.type === 'user/message').length).toBe(1)
  284. await run.dispose()
  285. })
  286. it('an errored child keeps its honest error result (no capture expected)', async () => {
  287. // Script exhaustion on the first call → the child turn errors.
  288. const { ctx, parent, adapter } = await setup([])
  289. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  290. const result = await run.result
  291. expect(result.stopReason).toBe('error')
  292. expect(adapter.requests.length).toBe(1)
  293. await run.dispose()
  294. })
  295. it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => {
  296. const { ctx, parent } = await setup([textResponse('prose, no capture')])
  297. const controller = new AbortController()
  298. const run = await ctx.subagents.start('spawn', structuredRequest(parent, { signal: controller.signal }))
  299. // Cancel synchronously inside the turn's end recording: the cancel
  300. // contract outranks the schema shortfall, so the result maps to aborted.
  301. ctx.on('session/event', (session, event) => {
  302. const child = ctx.agents.get(run.id)
  303. if (session === child?.session && event.type === 'turn/end') controller.abort('cancelled at turn end')
  304. })
  305. const result = await run.result
  306. expect(result.stopReason).toBe('aborted')
  307. await run.dispose()
  308. })
  309. it('rejects a schema outside the subset loud, before any child exists', async () => {
  310. const { ctx, parent } = await setup([])
  311. await expect(ctx.subagents.start('spawn', structuredRequest(parent, {
  312. outputSchema: { type: 'object', oneOf: [] } as unknown as StructuredOutputSchema,
  313. }))).rejects.toThrow(/unsupported output schema/)
  314. expect(ctx.agents.get(AgentId('parent'))).toBeDefined()
  315. })
  316. it('a schema carrying non-JSON values fails as OutputSchemaError at the validation boundary', async () => {
  317. const { ctx, parent } = await setup([])
  318. // Semantic assertion runs before provider startup.
  319. await expect(ctx.subagents.start('spawn', structuredRequest(parent, {
  320. outputSchema: { type: 'object', default: () => {} } as unknown as StructuredOutputSchema,
  321. }))).rejects.toThrow(/unsupported output schema.*annotation must be JSON data/)
  322. })
  323. it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => {
  324. const { ctx, parent, adapter } = await setup([
  325. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  326. textResponse('continues after the blocked capture'),
  327. ])
  328. // A PostToolUse-style hook turns the tool body's provisional success into
  329. // the authoritative final error observed by the commit notification.
  330. ctx.on('tools/post-execute', (exec, _result, next) => {
  331. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  332. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] })
  333. }
  334. return next()
  335. })
  336. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  337. const result = await run.result
  338. // No capture was committed: the run reports the schema shortfall...
  339. expect(result.structured).toBeUndefined()
  340. expect(result.stopReason).toBe('error')
  341. // ...the logged tool result is the blocked isError with the feedback...
  342. const child = ctx.agents.get(run.id)!
  343. const results = child.session.events.filter(e => e.type === 'tool/result')
  344. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  345. expect(JSON.stringify((results[0]!.data as { content: unknown }).content)).toContain('capture rejected by hook')
  346. // ...and the turn CONTINUED past the blocked call (no captured veto):
  347. // the model got to react to the failure with a second step.
  348. expect(adapter.requests.length).toBe(2)
  349. await run.dispose()
  350. })
  351. it('a post-execute accept-with-replacement still commits the capture', async () => {
  352. const { ctx, parent } = await setup([
  353. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  354. ])
  355. ctx.on('tools/post-execute', (exec, _result, next) => {
  356. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  357. return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] })
  358. }
  359. return next()
  360. })
  361. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  362. const result = await run.result
  363. expect(result.stopReason).toBe('completed')
  364. expect(result.structured).toEqual({ answer: 8 })
  365. await run.dispose()
  366. })
  367. it('commits only after a later prepended post-execute wrapper returns the authoritative result', async () => {
  368. const { ctx, parent } = await setup([
  369. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  370. textResponse('capture was rejected'),
  371. ])
  372. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  373. // Registered after attachment and prepended, so it wraps every listener
  374. // the child installed. It delegates first, then converts the apparent
  375. // capture success into the pipeline's authoritative failure.
  376. ctx.on('tools/post-execute', async (exec, _result, next) => {
  377. const downstream = await next()
  378. if (exec.name !== STRUCTURED_OUTPUT_TOOL) return downstream
  379. return { kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected after downstream' }] }
  380. }, { prepend: true })
  381. const result = await run.result
  382. expect(result.structured).toBeUndefined()
  383. expect(result.stopReason).toBe('error')
  384. const child = ctx.agents.get(run.id)
  385. const captureResult = child?.session.events.find(event =>
  386. event.type === 'tool/result' && event.data.callId === 'c1')
  387. expect(captureResult?.type === 'tool/result' && captureResult.data.isError).toBe(true)
  388. await run.dispose()
  389. })
  390. it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => {
  391. const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })])
  392. // A context-wide section stands in for the deployment persona: the
  393. // instruction must APPEND to the other scoped and global sections, not
  394. // replace them (AgentOptions has no prompt field — the instruction is an
  395. // ordinary child-scoped prompt registration).
  396. ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' })
  397. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  398. await run.result
  399. const childRequest = adapter.requests.at(-1)!
  400. expect(childRequest.system).toContain('You are a counter.')
  401. expect(childRequest.system!.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  402. expect(childRequest.system!.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0)
  403. await run.dispose()
  404. })
  405. it('keeps pure Code Mode at one wire tool and exposes structured capture through the SDK only', async () => {
  406. const { ctx, parent, adapter } = await setup([
  407. toolCallResponse('c1', RUN_CODE_NAME, { code: 'return await tools.structured_output({ answer: 12 })' }),
  408. ], {
  409. toolMode: 'code',
  410. codeRun: async (request) => {
  411. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  412. if (!capture) throw new Error('structured_output binding missing')
  413. await capture({ answer: 12 })
  414. return { logs: [], value: 'captured' }
  415. },
  416. })
  417. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  418. const result = await run.result
  419. expect(result.structured).toEqual({ answer: 12 })
  420. const request = adapter.requests[0]!
  421. expect(toolNames(request)).toEqual([RUN_CODE_NAME])
  422. expect(request.system).toContain('declare const tools:')
  423. expect(request.system).toContain('structured_output(args:')
  424. expect(request.system).toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  425. await run.dispose()
  426. })
  427. it('discards a nested capture when the enclosing run_code execution fails', async () => {
  428. const { ctx, parent, adapter } = await setup([
  429. toolCallResponse('c1', RUN_CODE_NAME, { code: 'await tools.structured_output({ answer: 12 }); throw new Error("boom")' }),
  430. textResponse('outer code failed'),
  431. ], {
  432. toolMode: 'code',
  433. codeRun: async (request) => {
  434. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  435. if (!capture) throw new Error('structured_output binding missing')
  436. await capture({ answer: 12 })
  437. return {
  438. logs: [],
  439. error: { kind: 'runtime', message: 'boom after capture' },
  440. } as never
  441. },
  442. })
  443. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  444. const result = await run.result
  445. expect(result.structured).toBeUndefined()
  446. expect(result.stopReason).toBe('error')
  447. expect(adapter.requests).toHaveLength(2)
  448. const child = ctx.agents.get(run.id)!
  449. const outer = child.session.events.find(event =>
  450. event.type === 'tool/result' && event.data.callId === CallId('c1'))
  451. expect(outer?.type === 'tool/result' && outer.data.isError).toBe(true)
  452. await run.dispose()
  453. })
  454. it('discards a nested capture when post-policy blocks the enclosing run_code result', async () => {
  455. const { ctx, parent, adapter } = await setup([
  456. toolCallResponse('c1', RUN_CODE_NAME, { code: 'return await tools.structured_output({ answer: 12 })' }),
  457. textResponse('outer code was blocked'),
  458. ], {
  459. toolMode: 'code',
  460. codeRun: async (request) => {
  461. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  462. if (!capture) throw new Error('structured_output binding missing')
  463. await capture({ answer: 12 })
  464. return { logs: [], value: 'captured' }
  465. },
  466. })
  467. ctx.on('tools/post-execute', (exec, _result, next) => exec.name === RUN_CODE_NAME
  468. ? Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'outer blocked' }] })
  469. : next())
  470. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  471. const result = await run.result
  472. expect(result.structured).toBeUndefined()
  473. expect(result.stopReason).toBe('error')
  474. expect(adapter.requests).toHaveLength(2)
  475. await run.dispose()
  476. })
  477. it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => {
  478. const { ctx, parent, adapter } = await setup([
  479. textResponse('parent answer'),
  480. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  481. ])
  482. parent.send([{ type: 'text', text: 'hello' }])
  483. await parent.whenIdle()
  484. expect(adapter.requests[0]!.system ?? '').not.toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  485. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  486. await run.result
  487. // The loop always assembles a base prompt (the harness identity section),
  488. // so the instruction APPENDS — never replaces.
  489. const childSystem = adapter.requests.at(-1)!.system!
  490. expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  491. expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length)
  492. await run.dispose()
  493. })
  494. describe('scoped registration (each child owns its capture tool)', () => {
  495. it('a plain agent never sees the tool: nothing is registered globally at all', async () => {
  496. const { ctx, parent, adapter } = await setup([textResponse('parent answer')])
  497. parent.send([{ type: 'text', text: 'hello' }])
  498. await parent.whenIdle()
  499. // Scoped registration: the global view has no capture tool, ever.
  500. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  501. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  502. })
  503. it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => {
  504. const { ctx, parent, adapter } = await setup([
  505. // Parent turn (a plain agent): must NOT see the tool.
  506. textResponse('parent answer'),
  507. // Child turn: must see it, with the run's schema.
  508. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
  509. ])
  510. parent.send([{ type: 'text', text: 'hello' }])
  511. await parent.whenIdle()
  512. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  513. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  514. await run.result
  515. const childRequest = adapter.requests[1]!
  516. expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL)
  517. const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  518. expect(entry.parameters).toEqual(SCHEMA)
  519. await run.dispose()
  520. })
  521. it('two concurrent structured children each see their OWN schema', async () => {
  522. const otherSchema: StructuredOutputSchema = {
  523. type: 'object',
  524. properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } },
  525. required: ['verdict'],
  526. }
  527. const { ctx, parent, adapter } = await setup([
  528. (options: GenerateOptions) => {
  529. // Answer with whatever schema this child was given — proves each
  530. // request carried the right one regardless of scheduling order.
  531. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  532. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  533. ? { verdict: 'real' }
  534. : { answer: 1 }
  535. return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args)
  536. },
  537. (options: GenerateOptions) => {
  538. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  539. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  540. ? { verdict: 'real' }
  541. : { answer: 1 }
  542. return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args)
  543. },
  544. ])
  545. const runA = await ctx.subagents.start('spawn', structuredRequest(parent))
  546. const runB = await ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema }))
  547. const [a, b] = await Promise.all([runA.result, runB.result])
  548. expect(a.structured).toEqual({ answer: 1 })
  549. expect(b.structured).toEqual({ verdict: 'real' })
  550. const schemas = adapter.requests.map(request =>
  551. request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters)
  552. expect(schemas).toContainEqual(SCHEMA)
  553. expect(schemas).toContainEqual(otherSchema)
  554. await runA.dispose()
  555. await runB.dispose()
  556. })
  557. it('places the capture tool and instruction in their canonical orders', async () => {
  558. const { ctx, parent, adapter } = await setup([
  559. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  560. ])
  561. // A global tool sorts lexicographically after structured_output, while a
  562. // global section above the 190 band follows the capture instruction.
  563. ctx.tools.register({
  564. name: 'zz_probe',
  565. description: 'probe',
  566. parameters: { type: 'object', properties: {} },
  567. execute: () => Promise.resolve([{ type: 'text', text: 'x' }]),
  568. })
  569. ctx.systemPrompt.section({ name: 'after-band', order: 200, text: 'AFTER-BAND' })
  570. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  571. await run.result
  572. const request = adapter.requests[0]!
  573. const names = toolNames(request)
  574. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeGreaterThanOrEqual(0)
  575. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeLessThan(names.indexOf('zz_probe'))
  576. const system = request.system ?? ''
  577. const instructionAt = system.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)
  578. expect(instructionAt).toBeGreaterThanOrEqual(0)
  579. expect(system.indexOf('AFTER-BAND')).toBeGreaterThan(instructionAt)
  580. await run.dispose()
  581. })
  582. it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => {
  583. const { parent, adapter } = await setup([textResponse('plain')])
  584. parent.send([{ type: 'text', text: 'q' }])
  585. await parent.whenIdle()
  586. const request = adapter.requests[0]!
  587. expect(request.tools).toBeUndefined()
  588. await new Promise(resolve => setTimeout(resolve, 0))
  589. })
  590. it('registrations ride the child fiber: disposing the run removes them; a provider reload mid-run cannot', async () => {
  591. const { ctx, parent, disposeProvider } = await setup([
  592. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }),
  593. ])
  594. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  595. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  596. // A backend hot-reload mid-run must not unregister the capture tool out
  597. // from under the live child: the registration rides the CHILD's fiber.
  598. disposeProvider()
  599. const result = await run.result
  600. expect(result.structured).toEqual({ answer: 4 })
  601. const child = ctx.agents.get(run.id)!
  602. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeDefined()
  603. await run.dispose()
  604. // Child disposed ⇒ its scoped registrations are gone.
  605. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeUndefined()
  606. })
  607. })
  608. it('a structured_output call from an agent WITHOUT a structured run is UNKNOWN_TOOL (the tool does not exist for it)', async () => {
  609. const { ctx, parent } = await setup([])
  610. const result = await ctx.tools.execute({
  611. callId: 'x' as never,
  612. name: STRUCTURED_OUTPUT_TOOL,
  613. arguments: { answer: 1 },
  614. agent: parent,
  615. })
  616. expect(result.isError).toBe(true)
  617. expect(result.error?.code).toBe('UNKNOWN_TOOL')
  618. })
  619. it('a structured_output call with NO calling agent at all is UNKNOWN_TOOL', async () => {
  620. const { ctx } = await setup([])
  621. const result = await ctx.tools.execute({
  622. callId: 'x' as never,
  623. name: STRUCTURED_OUTPUT_TOOL,
  624. arguments: { answer: 1 },
  625. })
  626. expect(result.isError).toBe(true)
  627. expect(result.error?.code).toBe('UNKNOWN_TOOL')
  628. })
  629. it('a failed execution stage is discarded and never promoted by a later call', async () => {
  630. const { ctx, parent } = await setup([
  631. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  632. ])
  633. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  634. // A prepended post-execute listener blocks the first capture without
  635. // delegating. The final-result notification discards that execution's
  636. // stage when it observes the error.
  637. let blocks = 1
  638. ctx.on('tools/post-execute', (exec, _result, next) => {
  639. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  640. blocks -= 1
  641. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  642. }
  643. return next()
  644. }, { prepend: true })
  645. const result = await run.result
  646. const child = ctx.agents.get(run.id)!
  647. // The blocked capture must NOT surface as structured success…
  648. expect(result.stopReason).toBe('error')
  649. expect(result.structured).toBeUndefined()
  650. // …and a LATER invalid call (its own body staged nothing) must not
  651. // resurrect c1's discarded value: drive the pipeline directly.
  652. const invalid = await ctx.tools.execute({
  653. callId: 'c2' as never,
  654. name: STRUCTURED_OUTPUT_TOOL,
  655. arguments: { answer: 'not-a-number' },
  656. agent: child,
  657. })
  658. expect(invalid.isError).toBe(true)
  659. // A fresh valid call still captures ITS OWN value.
  660. const valid = await ctx.tools.execute({
  661. callId: 'c3' as never,
  662. name: STRUCTURED_OUTPUT_TOOL,
  663. arguments: { answer: 9 },
  664. agent: child,
  665. })
  666. expect(valid.isError).toBeFalsy()
  667. await run.dispose()
  668. })
  669. it('reusing a failed execution\'s call id never promotes its discarded stage', async () => {
  670. const { ctx, parent } = await setup([
  671. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  672. ])
  673. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  674. // Block the first capture after its body stages a value. Its final error
  675. // discards that execution's stage.
  676. let blocks = 1
  677. ctx.on('tools/post-execute', (exec, _result, next) => {
  678. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  679. blocks -= 1
  680. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  681. }
  682. return next()
  683. }, { prepend: true })
  684. await run.result
  685. const child = ctx.agents.get(run.id)!
  686. // A SECOND capture call with the SAME call id whose body never stages
  687. // (invalid args throw before the stage): the discarded value must not ride
  688. // its acceptance.
  689. const reused = await ctx.tools.execute({
  690. callId: 'c1' as never,
  691. name: STRUCTURED_OUTPUT_TOOL,
  692. arguments: { answer: 'not-a-number' },
  693. agent: child,
  694. })
  695. expect(reused.isError).toBe(true)
  696. // Nothing was ever committed: a fresh valid call is still required.
  697. const valid = await ctx.tools.execute({
  698. callId: 'c1' as never,
  699. name: STRUCTURED_OUTPUT_TOOL,
  700. arguments: { answer: 5 },
  701. agent: child,
  702. })
  703. expect(valid.isError).toBeFalsy()
  704. await run.dispose()
  705. })
  706. it('a pre-execute deny with call-id reuse cannot promote another execution\'s stage', async () => {
  707. const { ctx, parent } = await setup([
  708. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  709. ])
  710. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  711. // Discard the first capture's stage via a final post-execute block.
  712. let blocks = 1
  713. ctx.on('tools/post-execute', (exec, _result, next) => {
  714. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  715. blocks -= 1
  716. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  717. }
  718. return next()
  719. }, { prepend: true })
  720. await run.result
  721. const child = ctx.agents.get(run.id)!
  722. // A prepended pre-execute deny skips the body, while the denied call still
  723. // reaches the final notification with the same adapter-minted call id.
  724. const offDeny = ctx.on('tools/pre-execute', (exec) => {
  725. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  726. return Promise.resolve({ kind: 'deny' as const, reason: 'outer veto' })
  727. }
  728. return undefined as never
  729. }, { prepend: true })
  730. const denied = await ctx.tools.execute({
  731. callId: 'c1' as never,
  732. name: STRUCTURED_OUTPUT_TOOL,
  733. arguments: { answer: 2 },
  734. agent: child,
  735. })
  736. expect(denied.isError).toBe(true)
  737. offDeny()
  738. // The discarded value was never promoted: a fresh valid call is required
  739. // (and succeeds, proving the runtime is not wedged).
  740. const valid = await ctx.tools.execute({
  741. callId: 'c1' as never,
  742. name: STRUCTURED_OUTPUT_TOOL,
  743. arguments: { answer: 5 },
  744. agent: child,
  745. })
  746. expect(valid.isError).toBeFalsy()
  747. await run.dispose()
  748. })
  749. })