structured.spec.ts 36 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from 'cordis'
  3. import { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm'
  4. import type { ContinuationDecision } from '@deepseek-ai/dsh-agent'
  5. import { SessionId } from '@deepseek-ai/dsh-session'
  6. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  7. import { mountAgentLoopTestDependencies } from '@deepseek-ai/dsh-agent-loop-testkit'
  8. import InvariantService from '@deepseek-ai/dsh-invariants'
  9. import * as SessionInvariant from '@deepseek-ai/dsh-session/invariant'
  10. import * as AgentInvariant from '@deepseek-ai/dsh-agent/invariant'
  11. import * as AgentLoopInvariant from '@deepseek-ai/dsh-agent-loop/invariant'
  12. import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent'
  13. import type { Config as ToolConfig, StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
  14. import { RUN_CODE_NAME } from '@deepseek-ai/dsh-tools'
  15. import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
  16. import { startInProcessRun } from '../src/index.ts'
  17. import {
  18. STRUCTURED_OUTPUT_INSTRUCTION,
  19. STRUCTURED_OUTPUT_TOOL,
  20. } from '../src/structured.ts'
  21. type Script = ConstructorParameters<typeof MockAdapter>[0]
  22. async function mountInvariants(ctx: Context): Promise<void> {
  23. await ctx.plugin(InvariantService)
  24. await ctx.plugin(SessionInvariant)
  25. await ctx.plugin(AgentInvariant)
  26. await ctx.plugin(AgentLoopInvariant)
  27. }
  28. interface CodeRunRequestLike {
  29. bindings: { global: string; functions: Record<string, (args: unknown) => Promise<unknown>> }[]
  30. }
  31. interface SetupOptions {
  32. toolMode?: ToolConfig['mode']
  33. codeRun?: (request: CodeRunRequestLike) => Promise<{ logs: never[]; value?: unknown }>
  34. }
  35. const SCHEMA: StructuredOutputSchema = {
  36. type: 'object',
  37. properties: { answer: { type: 'number' }, note: { type: 'string' } },
  38. required: ['answer'],
  39. }
  40. /**
  41. * Real loop, scripted model, and inline fresh-conversation provider over the shared driver. Loading
  42. * spawn/fork here would create a dev-dependency cycle; their specs cover plugin integration while
  43. * this fixture isolates driver behavior and scripts the child's `structured_output` calls.
  44. */
  45. async function setup(script: Script, options: SetupOptions = {}) {
  46. const ctx = new Context()
  47. const adapter = new MockAdapter(script)
  48. await mountAgentLoopTestDependencies(ctx, {
  49. tools: { mode: options.toolMode ?? 'native' },
  50. })
  51. if (options.toolMode === 'code' || options.toolMode === 'both') {
  52. ctx.provide('codeRuntime', {
  53. language: 'typescript',
  54. isolation: 'test',
  55. run: options.codeRun ?? (() => Promise.resolve({ logs: [] })),
  56. } as never)
  57. }
  58. await mountInvariants(ctx)
  59. await ctx.plugin(AgentLoop, { agents: [] })
  60. await ctx.plugin(SubagentService)
  61. const disposeProvider = ctx.subagents.registerProvider({
  62. name: 'spawn',
  63. capabilities: { outputSchema: true, depthLimit: true, toolFilter: false, persona: false },
  64. inheritsParentContext: false,
  65. start: (request: SubagentStartRequest) => startInProcessRun(request, {}),
  66. })
  67. ctx.llm.registerAdapter(['mock'], adapter)
  68. const parent = ctx.agentLoop.create(SessionId('parent'), { provider: 'mock', model: 'mock' })
  69. return { ctx, parent, adapter, disposeProvider }
  70. }
  71. function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial<SubagentStartRequest>): SubagentStartRequest {
  72. return {
  73. prompt: [{ type: 'text', text: 'produce the answer' }],
  74. parent,
  75. signal: new AbortController().signal,
  76. outputSchema: SCHEMA,
  77. ...extra,
  78. }
  79. }
  80. /** The tool names of one recorded model request. */
  81. function toolNames(request: GenerateOptions): string[] {
  82. return (request.tools ?? []).map(tool => tool.name)
  83. }
  84. describe('in-process structured output', () => {
  85. it('captures a valid structured_output call and surfaces result.structured', async () => {
  86. const { ctx, parent } = await setup([
  87. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }),
  88. ])
  89. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  90. const result = await run.result
  91. expect(result.stopReason).toBe('completed')
  92. expect(result.structured).toEqual({ answer: 42, note: 'done' })
  93. await run.dispose()
  94. })
  95. it('stops the turn after a successful capture — no extra model step is spent', async () => {
  96. const { ctx, parent, adapter } = await setup([
  97. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  98. textResponse('MUST NOT BE CONSUMED'),
  99. ])
  100. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  101. await run.result
  102. // Default continuation would run a second step after the tool call; the
  103. // structured runtime's turn-continuation veto stops the turn instead.
  104. expect(adapter.requests.length).toBe(1)
  105. await run.dispose()
  106. })
  107. it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => {
  108. // One model response carrying structured_output FIRST and a side-effecting
  109. // call after it: the continuation veto only fires at step end, so without
  110. // the pre-execute deny the trailing call would still run after the final
  111. // answer was accepted.
  112. const response = [
  113. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  114. { type: 'block-start', index: 1, blockType: 'tool-call' },
  115. { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
  116. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  117. { type: 'finish', reason: { kind: 'tool-calls' } },
  118. ] as Script[number]
  119. const { ctx, parent } = await setup([response])
  120. let sideEffectRan = false
  121. ctx.tools.register({
  122. name: 'side_effect',
  123. description: 'probe',
  124. parameters: { type: 'object', properties: {} },
  125. execute(): Promise<ContentBlock[]> {
  126. sideEffectRan = true
  127. return Promise.resolve([{ type: 'text', text: 'ran' }])
  128. },
  129. })
  130. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  131. const result = await run.result
  132. expect(result.stopReason).toBe('completed')
  133. expect(result.structured).toEqual({ answer: 5 })
  134. // The deny skipped dispatch entirely: the probe body never ran.
  135. expect(sideEffectRan).toBe(false)
  136. await run.dispose()
  137. })
  138. it('a later prepended pre-execute listener cannot resurrect dispatch after capture', async () => {
  139. const response = [
  140. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  141. { type: 'block-start', index: 1, blockType: 'tool-call' },
  142. { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
  143. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  144. { type: 'finish', reason: { kind: 'tool-calls' } },
  145. ] as Script[number]
  146. const { ctx, parent } = await setup([response])
  147. let sideEffectRan = false
  148. ctx.tools.register({
  149. name: 'side_effect',
  150. description: 'probe',
  151. parameters: { type: 'object', properties: {} },
  152. execute(): Promise<ContentBlock[]> {
  153. sideEffectRan = true
  154. return Promise.resolve([{ type: 'text', text: 'ran' }])
  155. },
  156. })
  157. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  158. // Registered after the child and prepended: this listener returns allow
  159. // after every downstream pre-execute decision. The service-owned guard
  160. // runs after the waterfall and can only deny, so the body still cannot run.
  161. ctx.on('tools/pre-execute', async (_exec, next) => {
  162. await next()
  163. return { kind: 'allow' as const }
  164. }, { prepend: true })
  165. const result = await run.result
  166. expect(result.structured).toEqual({ answer: 5 })
  167. expect(sideEffectRan).toBe(false)
  168. const child = ctx.agents.get(run.id)
  169. const sideEffectResult = child?.session.events.find(event =>
  170. event.type === 'tool/result' && event.data.callId === 'c2')
  171. expect(sideEffectResult?.type === 'tool/result' && sideEffectResult.data.isError).toBe(true)
  172. await run.dispose()
  173. })
  174. it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => {
  175. const response = [
  176. { type: 'block-start', index: 0, blockType: 'tool-call' },
  177. { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'side_effect', arguments: '{}' } },
  178. ...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk =>
  179. 'index' in chunk ? { ...chunk, index: 1 } : chunk),
  180. ] as Script[number]
  181. const { ctx, parent } = await setup([response])
  182. let sideEffectRan = false
  183. ctx.tools.register({
  184. name: 'side_effect',
  185. description: 'probe',
  186. parameters: { type: 'object', properties: {} },
  187. execute(): Promise<ContentBlock[]> {
  188. sideEffectRan = true
  189. return Promise.resolve([{ type: 'text', text: 'ran' }])
  190. },
  191. })
  192. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  193. const result = await run.result
  194. // The call ran BEFORE captured was set: the deny gate only guards the
  195. // window after the terminal answer landed.
  196. expect(sideEffectRan).toBe(true)
  197. expect(result.structured).toEqual({ answer: 6 })
  198. await run.dispose()
  199. })
  200. it('a later-prepended continuation wrapper cannot resurrect a captured turn', async () => {
  201. const { ctx, parent, adapter } = await setup([
  202. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  203. textResponse('MUST NOT BE CONSUMED'),
  204. ])
  205. ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'stop' }))
  206. let wrapperInstalled = false
  207. // Register before ready-only start: structured output is attached before session-start and the
  208. // loop. The wrapper waits for a downstream stop, rewrites it to continue, and must still lose
  209. // to the later terminal checkpoint.
  210. ctx.on('agent/session-start', (child) => {
  211. if (child === parent) return
  212. wrapperInstalled = true
  213. child.ctx.on('agent/turn-continuation', async (_subject, _turn, _decision, next): Promise<ContinuationDecision> => {
  214. const downstream = await next()
  215. expect(downstream).toEqual({ action: 'stop' })
  216. return { action: 'continue' }
  217. }, { prepend: true })
  218. })
  219. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  220. const result = await run.result
  221. expect(wrapperInstalled).toBe(true)
  222. expect(result.structured).toEqual({ answer: 7 })
  223. expect(result.stopReason).toBe('completed')
  224. expect(adapter.requests).toHaveLength(1)
  225. await run.dispose()
  226. })
  227. it('a continuation wrapper cannot carry steering past a captured terminal stop', async () => {
  228. const { ctx, parent, adapter } = await setup([
  229. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 9 }),
  230. textResponse('MUST NOT BE CONSUMED'),
  231. ])
  232. // A downstream policy stops, then a later wrapper delegates and queues steering that ordinary
  233. // folding would turn into continue. The terminal checkpoint must discard that steering.
  234. ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'stop' }))
  235. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  236. ctx.on('agent/session-start', (child) => {
  237. if (child.id !== run.id) return
  238. child.ctx.on('agent/turn-continuation', async (subject, _turn, _decision, next): Promise<ContinuationDecision> => {
  239. const downstream = await next()
  240. expect(downstream).toEqual({ action: 'stop' })
  241. subject.steer([{ type: 'text', text: 'late steering after downstream stop' }])
  242. return downstream
  243. }, { prepend: true })
  244. })
  245. const result = await run.result
  246. const child = ctx.agents.get(run.id)
  247. expect(result.structured).toEqual({ answer: 9 })
  248. expect(adapter.requests).toHaveLength(1)
  249. expect(child?.session.events.filter(event => event.type === 'turn/start')).toHaveLength(1)
  250. expect(child?.session.events.filter(event => event.type === 'steering/message')).toHaveLength(0)
  251. await run.dispose()
  252. })
  253. it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => {
  254. const { ctx, parent } = await setup([
  255. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }),
  256. toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  257. ])
  258. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  259. const result = await run.result
  260. expect(result.structured).toEqual({ answer: 7 })
  261. expect(result.stopReason).toBe('completed')
  262. // The child's log carries the isError tool/result for the invalid call.
  263. const child = ctx.agents.get(run.id)!
  264. const results = child.session.events.filter(e => e.type === 'tool/result')
  265. expect(results.length).toBe(2)
  266. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  267. await run.dispose()
  268. })
  269. it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => {
  270. const { ctx, parent, adapter } = await setup([
  271. textResponse('here is my answer in prose'),
  272. textResponse('MUST NOT BE CONSUMED'),
  273. ])
  274. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  275. const result = await run.result
  276. expect(result.stopReason).toBe('error')
  277. expect(result.structured).toBeUndefined()
  278. // Exactly one model request and one user message: no nudge turn exists.
  279. expect(adapter.requests.length).toBe(1)
  280. const child = ctx.agents.get(run.id)!
  281. expect(child.session.events.filter(e => e.type === 'user/message').length).toBe(1)
  282. await run.dispose()
  283. })
  284. it('an errored child keeps its honest error result (no capture expected)', async () => {
  285. // Script exhaustion on the first call → the child turn errors.
  286. const { ctx, parent, adapter } = await setup([])
  287. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  288. const result = await run.result
  289. expect(result.stopReason).toBe('error')
  290. expect(adapter.requests.length).toBe(1)
  291. await run.dispose()
  292. })
  293. it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => {
  294. const { ctx, parent } = await setup([textResponse('prose, no capture')])
  295. const controller = new AbortController()
  296. const run = await ctx.subagents.start('spawn', structuredRequest(parent, { signal: controller.signal }))
  297. // Cancel synchronously inside the turn's end recording: the cancel
  298. // contract outranks the schema shortfall, so the result maps to aborted.
  299. ctx.on('session/event', (session, event) => {
  300. const child = ctx.agents.get(run.id)
  301. if (session === child?.session && event.type === 'turn/end') controller.abort('cancelled at turn end')
  302. })
  303. const result = await run.result
  304. expect(result.stopReason).toBe('aborted')
  305. await run.dispose()
  306. })
  307. it('rejects a schema outside the subset loud, before any child exists', async () => {
  308. const { ctx, parent } = await setup([])
  309. await expect(ctx.subagents.start('spawn', structuredRequest(parent, {
  310. outputSchema: { type: 'object', oneOf: [] } as unknown as StructuredOutputSchema,
  311. }))).rejects.toThrow(/unsupported output schema/)
  312. expect(ctx.agents.get(SessionId('parent'))).toBeDefined()
  313. })
  314. it('a schema carrying non-JSON values fails as OutputSchemaError at the validation boundary', async () => {
  315. const { ctx, parent } = await setup([])
  316. // Semantic assertion runs before provider startup.
  317. await expect(ctx.subagents.start('spawn', structuredRequest(parent, {
  318. outputSchema: { type: 'object', default: () => {} } as unknown as StructuredOutputSchema,
  319. }))).rejects.toThrow(/unsupported output schema.*annotation must be JSON data/)
  320. })
  321. it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => {
  322. const { ctx, parent, adapter } = await setup([
  323. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  324. textResponse('continues after the blocked capture'),
  325. ])
  326. // A PostToolUse-style hook turns the tool body's provisional success into
  327. // the authoritative final error observed by the commit notification.
  328. ctx.on('tools/post-execute', (exec, _result, next) => {
  329. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  330. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] })
  331. }
  332. return next()
  333. })
  334. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  335. const result = await run.result
  336. // No capture was committed: the run reports the schema shortfall...
  337. expect(result.structured).toBeUndefined()
  338. expect(result.stopReason).toBe('error')
  339. // ...the logged tool result is the blocked isError with the feedback...
  340. const child = ctx.agents.get(run.id)!
  341. const results = child.session.events.filter(e => e.type === 'tool/result')
  342. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  343. expect(JSON.stringify((results[0]!.data as { content: unknown }).content)).toContain('capture rejected by hook')
  344. // ...and the turn CONTINUED past the blocked call (no captured veto):
  345. // the model got to react to the failure with a second step.
  346. expect(adapter.requests.length).toBe(2)
  347. await run.dispose()
  348. })
  349. it('a post-execute accept-with-replacement still commits the capture', async () => {
  350. const { ctx, parent } = await setup([
  351. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  352. ])
  353. ctx.on('tools/post-execute', (exec, _result, next) => {
  354. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  355. return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] })
  356. }
  357. return next()
  358. })
  359. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  360. const result = await run.result
  361. expect(result.stopReason).toBe('completed')
  362. expect(result.structured).toEqual({ answer: 8 })
  363. await run.dispose()
  364. })
  365. it('commits only after a later prepended post-execute wrapper returns the authoritative result', async () => {
  366. const { ctx, parent } = await setup([
  367. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  368. textResponse('capture was rejected'),
  369. ])
  370. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  371. // Registered after attachment and prepended, so it wraps every listener
  372. // the child installed. It delegates first, then converts the apparent
  373. // capture success into the pipeline's authoritative failure.
  374. ctx.on('tools/post-execute', async (exec, _result, next) => {
  375. const downstream = await next()
  376. if (exec.name !== STRUCTURED_OUTPUT_TOOL) return downstream
  377. return { kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected after downstream' }] }
  378. }, { prepend: true })
  379. const result = await run.result
  380. expect(result.structured).toBeUndefined()
  381. expect(result.stopReason).toBe('error')
  382. const child = ctx.agents.get(run.id)
  383. const captureResult = child?.session.events.find(event =>
  384. event.type === 'tool/result' && event.data.callId === 'c1')
  385. expect(captureResult?.type === 'tool/result' && captureResult.data.isError).toBe(true)
  386. await run.dispose()
  387. })
  388. it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => {
  389. const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })])
  390. // A context-wide section stands in for the deployment persona: the
  391. // instruction must APPEND to the other scoped and global sections, not
  392. // replace them (AgentOptions has no prompt field — the instruction is an
  393. // ordinary child-scoped prompt registration).
  394. ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' })
  395. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  396. await run.result
  397. const childRequest = adapter.requests.at(-1)!
  398. expect(childRequest.system).toContain('You are a counter.')
  399. expect(childRequest.system!.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  400. expect(childRequest.system!.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0)
  401. await run.dispose()
  402. })
  403. it('keeps pure Code Mode at one wire tool and exposes structured capture through the SDK only', async () => {
  404. const { ctx, parent, adapter } = await setup([
  405. toolCallResponse('c1', RUN_CODE_NAME, { code: 'return await tools.structured_output({ answer: 12 })' }),
  406. ], {
  407. toolMode: 'code',
  408. codeRun: async (request) => {
  409. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  410. if (!capture) throw new Error('structured_output binding missing')
  411. await capture({ answer: 12 })
  412. return { logs: [], value: 'captured' }
  413. },
  414. })
  415. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  416. const result = await run.result
  417. expect(result.structured).toEqual({ answer: 12 })
  418. const request = adapter.requests[0]!
  419. expect(toolNames(request)).toEqual([RUN_CODE_NAME])
  420. expect(request.system).toContain('declare const tools:')
  421. expect(request.system).toContain('structured_output(args:')
  422. expect(request.system).toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  423. await run.dispose()
  424. })
  425. it('discards a nested capture when the enclosing run_code execution fails', async () => {
  426. const { ctx, parent, adapter } = await setup([
  427. toolCallResponse('c1', RUN_CODE_NAME, { code: 'await tools.structured_output({ answer: 12 }); throw new Error("boom")' }),
  428. textResponse('outer code failed'),
  429. ], {
  430. toolMode: 'code',
  431. codeRun: async (request) => {
  432. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  433. if (!capture) throw new Error('structured_output binding missing')
  434. await capture({ answer: 12 })
  435. return {
  436. logs: [],
  437. error: { kind: 'runtime', message: 'boom after capture' },
  438. } as never
  439. },
  440. })
  441. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  442. const result = await run.result
  443. expect(result.structured).toBeUndefined()
  444. expect(result.stopReason).toBe('error')
  445. expect(adapter.requests).toHaveLength(2)
  446. const child = ctx.agents.get(run.id)!
  447. const outer = child.session.events.find(event =>
  448. event.type === 'tool/result' && event.data.callId === CallId('c1'))
  449. expect(outer?.type === 'tool/result' && outer.data.isError).toBe(true)
  450. await run.dispose()
  451. })
  452. it('discards a nested capture when post-policy blocks the enclosing run_code result', async () => {
  453. const { ctx, parent, adapter } = await setup([
  454. toolCallResponse('c1', RUN_CODE_NAME, { code: 'return await tools.structured_output({ answer: 12 })' }),
  455. textResponse('outer code was blocked'),
  456. ], {
  457. toolMode: 'code',
  458. codeRun: async (request) => {
  459. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  460. if (!capture) throw new Error('structured_output binding missing')
  461. await capture({ answer: 12 })
  462. return { logs: [], value: 'captured' }
  463. },
  464. })
  465. ctx.on('tools/post-execute', (exec, _result, next) => exec.name === RUN_CODE_NAME
  466. ? Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'outer blocked' }] })
  467. : next())
  468. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  469. const result = await run.result
  470. expect(result.structured).toBeUndefined()
  471. expect(result.stopReason).toBe('error')
  472. expect(adapter.requests).toHaveLength(2)
  473. await run.dispose()
  474. })
  475. it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => {
  476. const { ctx, parent, adapter } = await setup([
  477. textResponse('parent answer'),
  478. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  479. ])
  480. parent.send([{ type: 'text', text: 'hello' }])
  481. await parent.whenIdle()
  482. expect(adapter.requests[0]!.system ?? '').not.toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  483. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  484. await run.result
  485. // The loop always assembles a base prompt (the harness identity section),
  486. // so the instruction APPENDS — never replaces.
  487. const childSystem = adapter.requests.at(-1)!.system!
  488. expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  489. expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length)
  490. await run.dispose()
  491. })
  492. describe('scoped registration (each child owns its capture tool)', () => {
  493. it('a plain agent never sees the tool: nothing is registered globally at all', async () => {
  494. const { ctx, parent, adapter } = await setup([textResponse('parent answer')])
  495. parent.send([{ type: 'text', text: 'hello' }])
  496. await parent.whenIdle()
  497. // Scoped registration: the global view has no capture tool, ever.
  498. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  499. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  500. })
  501. it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => {
  502. const { ctx, parent, adapter } = await setup([
  503. // Parent turn (a plain agent): must NOT see the tool.
  504. textResponse('parent answer'),
  505. // Child turn: must see it, with the run's schema.
  506. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
  507. ])
  508. parent.send([{ type: 'text', text: 'hello' }])
  509. await parent.whenIdle()
  510. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  511. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  512. await run.result
  513. const childRequest = adapter.requests[1]!
  514. expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL)
  515. const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  516. expect(entry.parameters).toEqual(SCHEMA)
  517. await run.dispose()
  518. })
  519. it('two concurrent structured children each see their OWN schema', async () => {
  520. const otherSchema: StructuredOutputSchema = {
  521. type: 'object',
  522. properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } },
  523. required: ['verdict'],
  524. }
  525. const { ctx, parent, adapter } = await setup([
  526. (options: GenerateOptions) => {
  527. // Answer with whatever schema this child was given — proves each
  528. // request carried the right one regardless of scheduling order.
  529. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  530. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  531. ? { verdict: 'real' }
  532. : { answer: 1 }
  533. return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args)
  534. },
  535. (options: GenerateOptions) => {
  536. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  537. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  538. ? { verdict: 'real' }
  539. : { answer: 1 }
  540. return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args)
  541. },
  542. ])
  543. const runA = await ctx.subagents.start('spawn', structuredRequest(parent))
  544. const runB = await ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema }))
  545. const [a, b] = await Promise.all([runA.result, runB.result])
  546. expect(a.structured).toEqual({ answer: 1 })
  547. expect(b.structured).toEqual({ verdict: 'real' })
  548. const schemas = adapter.requests.map(request =>
  549. request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters)
  550. expect(schemas).toContainEqual(SCHEMA)
  551. expect(schemas).toContainEqual(otherSchema)
  552. await runA.dispose()
  553. await runB.dispose()
  554. })
  555. it('places the capture tool and instruction in their canonical orders', async () => {
  556. const { ctx, parent, adapter } = await setup([
  557. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  558. ])
  559. // A global tool sorts lexicographically after structured_output, while a
  560. // global section above the 190 band follows the capture instruction.
  561. ctx.tools.register({
  562. name: 'zz_probe',
  563. description: 'probe',
  564. parameters: { type: 'object', properties: {} },
  565. execute: () => Promise.resolve([{ type: 'text', text: 'x' }]),
  566. })
  567. ctx.systemPrompt.section({ name: 'after-band', order: 200, text: 'AFTER-BAND' })
  568. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  569. await run.result
  570. const request = adapter.requests[0]!
  571. const names = toolNames(request)
  572. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeGreaterThanOrEqual(0)
  573. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeLessThan(names.indexOf('zz_probe'))
  574. const system = request.system ?? ''
  575. const instructionAt = system.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)
  576. expect(instructionAt).toBeGreaterThanOrEqual(0)
  577. expect(system.indexOf('AFTER-BAND')).toBeGreaterThan(instructionAt)
  578. await run.dispose()
  579. })
  580. it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => {
  581. const { parent, adapter } = await setup([textResponse('plain')])
  582. parent.send([{ type: 'text', text: 'q' }])
  583. await parent.whenIdle()
  584. const request = adapter.requests[0]!
  585. expect(request.tools).toBeUndefined()
  586. await new Promise(resolve => setTimeout(resolve, 0))
  587. })
  588. it('registrations ride the child fiber: disposing the run removes them; a provider reload mid-run cannot', async () => {
  589. const { ctx, parent, disposeProvider } = await setup([
  590. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }),
  591. ])
  592. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  593. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  594. // A backend hot-reload mid-run must not unregister the capture tool out
  595. // from under the live child: the registration rides the CHILD's fiber.
  596. disposeProvider()
  597. const result = await run.result
  598. expect(result.structured).toEqual({ answer: 4 })
  599. const child = ctx.agents.get(run.id)!
  600. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeDefined()
  601. await run.dispose()
  602. // Child disposed ⇒ its scoped registrations are gone.
  603. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeUndefined()
  604. })
  605. })
  606. it('a structured_output call from an agent WITHOUT a structured run is UNKNOWN_TOOL (the tool does not exist for it)', async () => {
  607. const { ctx, parent } = await setup([])
  608. const result = await ctx.tools.execute({
  609. callId: 'x' as never,
  610. name: STRUCTURED_OUTPUT_TOOL,
  611. arguments: { answer: 1 },
  612. agent: parent,
  613. })
  614. expect(result.isError).toBe(true)
  615. expect(result.error?.code).toBe('UNKNOWN_TOOL')
  616. })
  617. it('a structured_output call with NO calling agent at all is UNKNOWN_TOOL', async () => {
  618. const { ctx } = await setup([])
  619. const result = await ctx.tools.execute({
  620. callId: 'x' as never,
  621. name: STRUCTURED_OUTPUT_TOOL,
  622. arguments: { answer: 1 },
  623. })
  624. expect(result.isError).toBe(true)
  625. expect(result.error?.code).toBe('UNKNOWN_TOOL')
  626. })
  627. it('a failed execution stage is discarded and never promoted by a later call', async () => {
  628. const { ctx, parent } = await setup([
  629. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  630. ])
  631. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  632. // A prepended post-execute listener blocks the first capture without
  633. // delegating. The final-result notification discards that execution's
  634. // stage when it observes the error.
  635. let blocks = 1
  636. ctx.on('tools/post-execute', (exec, _result, next) => {
  637. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  638. blocks -= 1
  639. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  640. }
  641. return next()
  642. }, { prepend: true })
  643. const result = await run.result
  644. const child = ctx.agents.get(run.id)!
  645. // The blocked capture must NOT surface as structured success…
  646. expect(result.stopReason).toBe('error')
  647. expect(result.structured).toBeUndefined()
  648. // …and a LATER invalid call (its own body staged nothing) must not
  649. // resurrect c1's discarded value: drive the pipeline directly.
  650. const invalid = await ctx.tools.execute({
  651. callId: 'c2' as never,
  652. name: STRUCTURED_OUTPUT_TOOL,
  653. arguments: { answer: 'not-a-number' },
  654. agent: child,
  655. })
  656. expect(invalid.isError).toBe(true)
  657. // A fresh valid call still captures ITS OWN value.
  658. const valid = await ctx.tools.execute({
  659. callId: 'c3' as never,
  660. name: STRUCTURED_OUTPUT_TOOL,
  661. arguments: { answer: 9 },
  662. agent: child,
  663. })
  664. expect(valid.isError).toBeFalsy()
  665. await run.dispose()
  666. })
  667. it('reusing a failed execution\'s call id never promotes its discarded stage', async () => {
  668. const { ctx, parent } = await setup([
  669. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  670. ])
  671. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  672. // Block the first capture after its body stages a value. Its final error
  673. // discards that execution's stage.
  674. let blocks = 1
  675. ctx.on('tools/post-execute', (exec, _result, next) => {
  676. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  677. blocks -= 1
  678. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  679. }
  680. return next()
  681. }, { prepend: true })
  682. await run.result
  683. const child = ctx.agents.get(run.id)!
  684. // A SECOND capture call with the SAME call id whose body never stages
  685. // (invalid args throw before the stage): the discarded value must not ride
  686. // its acceptance.
  687. const reused = await ctx.tools.execute({
  688. callId: 'c1' as never,
  689. name: STRUCTURED_OUTPUT_TOOL,
  690. arguments: { answer: 'not-a-number' },
  691. agent: child,
  692. })
  693. expect(reused.isError).toBe(true)
  694. // Nothing was ever committed: a fresh valid call is still required.
  695. const valid = await ctx.tools.execute({
  696. callId: 'c1' as never,
  697. name: STRUCTURED_OUTPUT_TOOL,
  698. arguments: { answer: 5 },
  699. agent: child,
  700. })
  701. expect(valid.isError).toBeFalsy()
  702. await run.dispose()
  703. })
  704. it('a pre-execute deny with call-id reuse cannot promote another execution\'s stage', async () => {
  705. const { ctx, parent } = await setup([
  706. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  707. ])
  708. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  709. // Discard the first capture's stage via a final post-execute block.
  710. let blocks = 1
  711. ctx.on('tools/post-execute', (exec, _result, next) => {
  712. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  713. blocks -= 1
  714. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  715. }
  716. return next()
  717. }, { prepend: true })
  718. await run.result
  719. const child = ctx.agents.get(run.id)!
  720. // A prepended pre-execute deny skips the body, while the denied call still
  721. // reaches the final notification with the same adapter-minted call id.
  722. const offDeny = ctx.on('tools/pre-execute', (exec) => {
  723. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  724. return Promise.resolve({ kind: 'deny' as const, reason: 'outer veto' })
  725. }
  726. return undefined as never
  727. }, { prepend: true })
  728. const denied = await ctx.tools.execute({
  729. callId: 'c1' as never,
  730. name: STRUCTURED_OUTPUT_TOOL,
  731. arguments: { answer: 2 },
  732. agent: child,
  733. })
  734. expect(denied.isError).toBe(true)
  735. offDeny()
  736. // The discarded value was never promoted: a fresh valid call is required
  737. // (and succeeds, proving the runtime is not wedged).
  738. const valid = await ctx.tools.execute({
  739. callId: 'c1' as never,
  740. name: STRUCTURED_OUTPUT_TOOL,
  741. arguments: { answer: 5 },
  742. agent: child,
  743. })
  744. expect(valid.isError).toBeFalsy()
  745. await run.dispose()
  746. })
  747. })