1
0

structured.spec.ts 35 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from '@deepseek-ai/cordis'
  3. import { createUserMessage, ToolCallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm'
  4. import { SessionId } from '@deepseek-ai/dsh-session'
  5. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  6. import { mountAgentLoopTestDependencies } from '@deepseek-ai/dsh-agent-loop-testkit'
  7. import InvariantRegistry from '@deepseek-ai/dsh-invariants'
  8. import type {} from '@deepseek-ai/dsh-system-prompt'
  9. import * as SessionInvariant from '@deepseek-ai/dsh-session/invariant'
  10. import * as AgentInvariant from '@deepseek-ai/dsh-agent/invariant'
  11. import * as AgentLoopInvariant from '@deepseek-ai/dsh-agent-loop/invariant'
  12. import SubagentRuntime, {
  13. type ResolvedSubagentStartRequest,
  14. type SubagentStartRequest,
  15. } from '@deepseek-ai/dsh-subagent'
  16. import type { Config as ToolConfig, ObjectJsonSchema } from '@deepseek-ai/dsh-tools'
  17. import { defineContentToolFixture, RUN_CODE_NAME } from '@deepseek-ai/dsh-tools'
  18. import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
  19. import { startInProcessRun } from '../src/index.ts'
  20. import {
  21. STRUCTURED_OUTPUT_INSTRUCTION,
  22. STRUCTURED_OUTPUT_TOOL,
  23. } from '../src/structured.ts'
  24. const testToolSignal = new AbortController().signal
  25. type Script = ConstructorParameters<typeof MockAdapter>[0]
  26. async function mountInvariants(ctx: Context): Promise<void> {
  27. await ctx.plugin(InvariantRegistry)
  28. await ctx.plugin(SessionInvariant)
  29. await ctx.plugin(AgentInvariant)
  30. await ctx.plugin(AgentLoopInvariant)
  31. }
  32. interface PtcRunRequestLike {
  33. bindings: { global: string; functions: Record<string, (args: unknown) => Promise<unknown>> }[]
  34. }
  35. interface SetupOptions {
  36. toolMode?: ToolConfig['mode']
  37. codeRun?: (request: PtcRunRequestLike) => Promise<{ logs: never[]; value?: unknown }>
  38. }
  39. const SCHEMA: ObjectJsonSchema = {
  40. type: 'object',
  41. properties: { answer: { type: 'number' }, note: { type: 'string' } },
  42. required: ['answer'],
  43. }
  44. /**
  45. * Real loop, scripted model, and inline fresh-conversation provider over the shared driver. Loading
  46. * spawn/fork here would create a dev-dependency cycle; their specs cover plugin integration while
  47. * this fixture isolates driver behavior and scripts the child's `structured_output` calls.
  48. */
  49. async function setup(script: Script, options: SetupOptions = {}) {
  50. const ctx = new Context()
  51. const adapter = new MockAdapter(script)
  52. await mountAgentLoopTestDependencies(ctx, {
  53. tools: { mode: options.toolMode ?? 'native' },
  54. })
  55. if (options.toolMode === 'ptc' || options.toolMode === 'both') {
  56. ctx.provide('ptcRuntime', {
  57. language: 'typescript',
  58. isolation: 'test',
  59. resolve: (request: import('@deepseek-ai/dsh-ptc-runtime').PtcRunRequest) => ({ ...request, cwd: request.cwd ?? process.cwd(), timeoutMs: 120_000 }),
  60. run: options.codeRun ?? (() => Promise.resolve({ logs: [] })),
  61. } as never)
  62. }
  63. await mountInvariants(ctx)
  64. await ctx.plugin(AgentLoop, { agents: [] })
  65. await ctx.plugin(SubagentRuntime)
  66. const disposeProvider = ctx.subagents.registerProvider({
  67. name: 'spawn',
  68. capabilities: { agentOptions: true, outputSchema: true, depthLimit: true, toolFilter: false, persona: false },
  69. inheritsParentContext: false,
  70. start: (request: ResolvedSubagentStartRequest) => startInProcessRun(request, {}),
  71. })
  72. ctx.llm.registerAdapter(['mock'], adapter)
  73. const parent = await ctx.agentLoop.create(SessionId('parent'), { provider: 'mock', model: 'mock' })
  74. return { ctx, parent, adapter, disposeProvider }
  75. }
  76. function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial<SubagentStartRequest>): SubagentStartRequest {
  77. return {
  78. label: 'produce the answer',
  79. prompt: [{ type: 'text', text: 'produce the answer' }],
  80. parent,
  81. signal: new AbortController().signal,
  82. outputSchema: SCHEMA,
  83. ...extra,
  84. }
  85. }
  86. /** The tool names of one recorded model request. */
  87. function toolNames(request: GenerateOptions): string[] {
  88. return (request.tools ?? []).map(tool => tool.name)
  89. }
  90. /** Text of a loop-built request's leading system message; `''` when the request has none. */
  91. function requestSystem(request: GenerateOptions): string {
  92. const head = request.messages[0]
  93. if (head?.role !== 'system') return ''
  94. return head.content.filter(block => block.type === 'text').map(block => block.text).join('')
  95. }
  96. describe('in-process structured output', () => {
  97. it('captures a valid structured_output call and surfaces result.structured', async () => {
  98. const { ctx, parent } = await setup([
  99. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }),
  100. ])
  101. let acknowledgement: unknown
  102. ctx.on('tools/result', (exec, toolResult) => {
  103. if (exec.name === STRUCTURED_OUTPUT_TOOL && !toolResult.isError) acknowledgement = toolResult.value
  104. })
  105. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  106. const result = await run.result
  107. expect(result.stopReason).toBe('completed')
  108. expect(result.structured).toEqual({ answer: 42, note: 'done' })
  109. expect(acknowledgement).toEqual({ recorded: true })
  110. await run.dispose()
  111. })
  112. it('stops the turn after a successful capture — no extra model step is spent', async () => {
  113. const { ctx, parent, adapter } = await setup([
  114. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  115. textResponse('MUST NOT BE CONSUMED'),
  116. ])
  117. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  118. await run.result
  119. // The structured tool marks its successful result as turn-concluding.
  120. expect(adapter.requests.length).toBe(1)
  121. await run.dispose()
  122. })
  123. it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => {
  124. // One model response carrying structured_output FIRST and a side-effecting
  125. // call after it: the continuation veto only fires at step end, so without
  126. // the pre-execute deny the trailing call would still run after the final
  127. // answer was accepted.
  128. const response = [
  129. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  130. { type: 'block-start', index: 1, blockType: 'tool-call' },
  131. { type: 'block-end', index: 1, block: { type: 'tool-call', id: ToolCallId('c2'), name: 'side_effect', arguments: '{}' } },
  132. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  133. { type: 'finish', reason: { kind: 'tool-calls' } },
  134. ] as Script[number]
  135. const { ctx, parent } = await setup([response])
  136. let sideEffectRan = false
  137. ctx.tools.register(defineContentToolFixture({
  138. name: 'side_effect',
  139. description: 'probe',
  140. parameters: {},
  141. execute(): Promise<ContentBlock[]> {
  142. sideEffectRan = true
  143. return Promise.resolve([{ type: 'text', text: 'ran' }])
  144. },
  145. }))
  146. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  147. const result = await run.result
  148. expect(result.stopReason).toBe('completed')
  149. expect(result.structured).toEqual({ answer: 5 })
  150. // The deny skipped dispatch entirely: the probe body never ran.
  151. expect(sideEffectRan).toBe(false)
  152. await run.dispose()
  153. })
  154. it('a later prepended pre-execute listener cannot resurrect dispatch after capture', async () => {
  155. const response = [
  156. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  157. { type: 'block-start', index: 1, blockType: 'tool-call' },
  158. { type: 'block-end', index: 1, block: { type: 'tool-call', id: ToolCallId('c2'), name: 'side_effect', arguments: '{}' } },
  159. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  160. { type: 'finish', reason: { kind: 'tool-calls' } },
  161. ] as Script[number]
  162. const { ctx, parent } = await setup([response])
  163. let sideEffectRan = false
  164. ctx.tools.register(defineContentToolFixture({
  165. name: 'side_effect',
  166. description: 'probe',
  167. parameters: {},
  168. execute(): Promise<ContentBlock[]> {
  169. sideEffectRan = true
  170. return Promise.resolve([{ type: 'text', text: 'ran' }])
  171. },
  172. }))
  173. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  174. // Registered after the child and prepended: this listener returns allow
  175. // after every downstream pre-execute decision. The service-owned guard
  176. // runs after the waterfall and can only deny, so the body still cannot run.
  177. ctx.on('tools/pre-execute', async (_exec, next) => {
  178. await next()
  179. return { kind: 'allow' as const }
  180. }, { prepend: true })
  181. const result = await run.result
  182. expect(result.structured).toEqual({ answer: 5 })
  183. expect(sideEffectRan).toBe(false)
  184. const child = ctx.agents.get(run.id)
  185. const sideEffectResult = child?.session.snapshotEvents().find(event =>
  186. event.type === 'tool/result' && event.data.message.source.callId === 'c2')
  187. expect(sideEffectResult?.type === 'tool/result' && sideEffectResult.data.message.content[0].isError).toBe(true)
  188. await run.dispose()
  189. })
  190. it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => {
  191. const response = [
  192. { type: 'block-start', index: 0, blockType: 'tool-call' },
  193. { type: 'block-end', index: 0, block: { type: 'tool-call', id: ToolCallId('c1'), name: 'side_effect', arguments: '{}' } },
  194. ...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk =>
  195. 'index' in chunk ? { ...chunk, index: 1 } : chunk),
  196. ] as Script[number]
  197. const { ctx, parent } = await setup([response])
  198. let sideEffectRan = false
  199. ctx.tools.register(defineContentToolFixture({
  200. name: 'side_effect',
  201. description: 'probe',
  202. parameters: {},
  203. execute(): Promise<ContentBlock[]> {
  204. sideEffectRan = true
  205. return Promise.resolve([{ type: 'text', text: 'ran' }])
  206. },
  207. }))
  208. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  209. const result = await run.result
  210. // The call ran BEFORE captured was set: the deny gate only guards the
  211. // window after the terminal answer landed.
  212. expect(sideEffectRan).toBe(true)
  213. expect(result.structured).toEqual({ answer: 6 })
  214. await run.dispose()
  215. })
  216. it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => {
  217. const { ctx, parent } = await setup([
  218. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }),
  219. toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  220. ])
  221. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  222. const result = await run.result
  223. expect(result.structured).toEqual({ answer: 7 })
  224. expect(result.stopReason).toBe('completed')
  225. // The child's log carries the isError tool/result for the invalid call.
  226. const child = ctx.agents.get(run.id)!
  227. const results = child.session.snapshotEvents().filter(e => e.type === 'tool/result')
  228. expect(results.length).toBe(2)
  229. expect(results[0]!.data.message.content[0].isError).toBe(true)
  230. await run.dispose()
  231. })
  232. it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => {
  233. const { ctx, parent, adapter } = await setup([
  234. textResponse('here is my answer in prose'),
  235. textResponse('MUST NOT BE CONSUMED'),
  236. ])
  237. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  238. const result = await run.result
  239. expect(result.stopReason).toBe('error')
  240. expect(result.structured).toBeUndefined()
  241. // Exactly one model request and one caller-supplied user message: no nudge turn exists.
  242. expect(adapter.requests.length).toBe(1)
  243. const child = ctx.agents.get(run.id)!
  244. expect(child.session.snapshotEvents().filter(e => e.type === 'user/message' && e.data.source.kind !== 'plugin').length).toBe(1)
  245. await run.dispose()
  246. })
  247. it('an errored child keeps its honest error result (no capture expected)', async () => {
  248. // Script exhaustion on the first call → the child turn errors.
  249. const { ctx, parent, adapter } = await setup([])
  250. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  251. const result = await run.result
  252. expect(result.stopReason).toBe('error')
  253. expect(adapter.requests.length).toBe(1)
  254. await run.dispose()
  255. })
  256. it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => {
  257. const { ctx, parent } = await setup([textResponse('prose, no capture')])
  258. const controller = new AbortController()
  259. const run = await ctx.subagents.start('spawn', structuredRequest(parent, { signal: controller.signal }))
  260. // Cancel synchronously inside the turn's end recording: the cancel
  261. // contract outranks the schema shortfall, so the result maps to aborted.
  262. ctx.on('session/event', (session, event) => {
  263. const child = ctx.agents.get(run.id)
  264. if (session === child?.session && event.type === 'turn/end') controller.abort('cancelled at turn end')
  265. })
  266. const result = await run.result
  267. expect(result.stopReason).toBe('aborted')
  268. await run.dispose()
  269. })
  270. it('rejects a schema outside the subset loud, before any child exists', async () => {
  271. const { ctx, parent } = await setup([])
  272. await expect(ctx.subagents.start('spawn', structuredRequest(parent, {
  273. outputSchema: { type: 'object', oneOf: [] } as unknown as ObjectJsonSchema,
  274. }))).rejects.toThrow(/unsupported JSON schema/)
  275. expect(ctx.agents.get(SessionId('parent'))).toBeDefined()
  276. })
  277. it('a schema carrying non-JSON values fails as JsonSchemaError at the validation boundary', async () => {
  278. const { ctx, parent } = await setup([])
  279. // Semantic assertion runs before provider startup.
  280. await expect(ctx.subagents.start('spawn', structuredRequest(parent, {
  281. outputSchema: { type: 'object', default: () => {} } as unknown as ObjectJsonSchema,
  282. }))).rejects.toThrow(/unsupported JSON schema.*annotation must be lossless JSON data/)
  283. })
  284. it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => {
  285. const { ctx, parent, adapter } = await setup([
  286. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  287. textResponse('continues after the blocked capture'),
  288. ])
  289. // A PostToolUse-style hook turns the tool body's provisional success into
  290. // the authoritative final error observed by the commit notification.
  291. ctx.on('tools/post-execute', (exec, _result, next) => {
  292. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  293. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] })
  294. }
  295. return next()
  296. })
  297. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  298. const result = await run.result
  299. // No capture was committed: the run reports the schema shortfall...
  300. expect(result.structured).toBeUndefined()
  301. expect(result.stopReason).toBe('error')
  302. // ...the logged tool result is the blocked isError with the feedback...
  303. const child = ctx.agents.get(run.id)!
  304. const results = child.session.snapshotEvents().filter(e => e.type === 'tool/result')
  305. expect(results[0]!.data.message.content[0].isError).toBe(true)
  306. expect(JSON.stringify(results[0]!.data.message.content)).toContain('capture rejected by hook')
  307. // ...and the turn CONTINUED past the blocked call (no captured veto):
  308. // the model got to react to the failure with a second step.
  309. expect(adapter.requests.length).toBe(2)
  310. await run.dispose()
  311. })
  312. it('a post-execute accept-with-replacement still commits the capture', async () => {
  313. const { ctx, parent } = await setup([
  314. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  315. ])
  316. ctx.on('tools/post-execute', (exec, _result, next) => {
  317. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  318. return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] })
  319. }
  320. return next()
  321. })
  322. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  323. const result = await run.result
  324. expect(result.stopReason).toBe('completed')
  325. expect(result.structured).toEqual({ answer: 8 })
  326. await run.dispose()
  327. })
  328. it('commits only after a later prepended post-execute wrapper returns the authoritative result', async () => {
  329. const { ctx, parent } = await setup([
  330. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  331. textResponse('capture was rejected'),
  332. ])
  333. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  334. // Registered after attachment and prepended, so it wraps every listener
  335. // the child installed. It delegates first, then converts the apparent
  336. // capture success into the pipeline's authoritative failure.
  337. ctx.on('tools/post-execute', async (exec, _result, next) => {
  338. const downstream = await next()
  339. if (exec.name !== STRUCTURED_OUTPUT_TOOL) return downstream
  340. return { kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected after downstream' }] }
  341. }, { prepend: true })
  342. const result = await run.result
  343. expect(result.structured).toBeUndefined()
  344. expect(result.stopReason).toBe('error')
  345. const child = ctx.agents.get(run.id)
  346. const captureResult = child?.session.snapshotEvents().find(event =>
  347. event.type === 'tool/result' && event.data.message.source.callId === 'c1')
  348. expect(captureResult?.type === 'tool/result' && captureResult.data.message.content[0].isError).toBe(true)
  349. await run.dispose()
  350. })
  351. it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => {
  352. const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })])
  353. // A context-wide section stands in for the deployment persona: the
  354. // instruction must APPEND to the other scoped and global sections, not
  355. // replace them (AgentOptions has no prompt field — the instruction is an
  356. // ordinary child-scoped prompt registration).
  357. ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' })
  358. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  359. await run.result
  360. const childSystem = requestSystem(adapter.requests.at(-1)!)
  361. expect(childSystem).toContain('You are a counter.')
  362. expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  363. expect(childSystem.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0)
  364. await run.dispose()
  365. })
  366. it('keeps pure PTC mode at one wire tool and exposes structured capture through the SDK only', async () => {
  367. const { ctx, parent, adapter } = await setup([
  368. toolCallResponse('c1', RUN_CODE_NAME, { code: 'return await tools.structured_output({ answer: 12 })', description: 'Capture the structured answer' }),
  369. ], {
  370. toolMode: 'ptc',
  371. codeRun: async (request) => {
  372. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  373. if (!capture) throw new Error('structured_output binding missing')
  374. await capture({ answer: 12 })
  375. return { logs: [], value: 'captured' }
  376. },
  377. })
  378. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  379. const result = await run.result
  380. expect(result.structured).toEqual({ answer: 12 })
  381. const request = adapter.requests[0]!
  382. expect(toolNames(request)).toEqual([RUN_CODE_NAME])
  383. const system = requestSystem(request)
  384. expect(system).toContain('interface ToolArgsMap')
  385. expect(system).toContain('interface ToolOutputMap')
  386. expect(system).toContain('recorded: true;')
  387. expect(system).toContain('Promise<ToolOutputMap[K]>')
  388. expect(system).toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  389. await run.dispose()
  390. })
  391. it('discards a nested capture when the enclosing run_code execution fails', async () => {
  392. const { ctx, parent, adapter } = await setup([
  393. toolCallResponse('c1', RUN_CODE_NAME, { code: 'await tools.structured_output({ answer: 12 }); throw new Error("boom")', description: 'Capture then fail the program' }),
  394. textResponse('outer code failed'),
  395. ], {
  396. toolMode: 'ptc',
  397. codeRun: async (request) => {
  398. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  399. if (!capture) throw new Error('structured_output binding missing')
  400. await capture({ answer: 12 })
  401. return {
  402. logs: [],
  403. error: { kind: 'runtime', message: 'boom after capture' },
  404. } as never
  405. },
  406. })
  407. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  408. const result = await run.result
  409. expect(result.structured).toBeUndefined()
  410. expect(result.stopReason).toBe('error')
  411. expect(adapter.requests).toHaveLength(2)
  412. const child = ctx.agents.get(run.id)!
  413. const outer = child.session.snapshotEvents().find(event =>
  414. event.type === 'tool/result' && event.data.message.source.callId === ToolCallId('c1'))
  415. expect(outer?.type === 'tool/result' && outer.data.message.content[0].isError).toBe(true)
  416. await run.dispose()
  417. })
  418. it('discards a nested capture when post-policy blocks the enclosing run_code result', async () => {
  419. const { ctx, parent, adapter } = await setup([
  420. toolCallResponse('c1', RUN_CODE_NAME, { code: 'return await tools.structured_output({ answer: 12 })', description: 'Capture the structured answer' }),
  421. textResponse('outer code was blocked'),
  422. ], {
  423. toolMode: 'ptc',
  424. codeRun: async (request) => {
  425. const capture = request.bindings.at(0)?.functions[STRUCTURED_OUTPUT_TOOL]
  426. if (!capture) throw new Error('structured_output binding missing')
  427. await capture({ answer: 12 })
  428. return { logs: [], value: 'captured' }
  429. },
  430. })
  431. ctx.on('tools/post-execute', (exec, _result, next) => exec.name === RUN_CODE_NAME
  432. ? Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'outer blocked' }] })
  433. : next())
  434. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  435. const result = await run.result
  436. expect(result.structured).toBeUndefined()
  437. expect(result.stopReason).toBe('error')
  438. expect(adapter.requests).toHaveLength(2)
  439. await run.dispose()
  440. })
  441. it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => {
  442. const { ctx, parent, adapter } = await setup([
  443. textResponse('parent answer'),
  444. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  445. ])
  446. parent.followup(createUserMessage({ content: [{ type: 'text', text: 'hello' }], source: { kind: 'user' } }))
  447. await parent.whenIdle()
  448. expect(requestSystem(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  449. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  450. await run.result
  451. // The loop always assembles a base prompt (the harness identity section),
  452. // so the instruction APPENDS — never replaces.
  453. const childSystem = requestSystem(adapter.requests.at(-1)!)
  454. expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  455. expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length)
  456. await run.dispose()
  457. })
  458. describe('scoped registration (each child owns its capture tool)', () => {
  459. it('a plain agent never sees the tool: nothing is registered globally at all', async () => {
  460. const { ctx, parent, adapter } = await setup([textResponse('parent answer')])
  461. parent.followup(createUserMessage({ content: [{ type: 'text', text: 'hello' }], source: { kind: 'user' } }))
  462. await parent.whenIdle()
  463. // Scoped registration: the global view has no capture tool, ever.
  464. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  465. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  466. })
  467. it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => {
  468. const { ctx, parent, adapter } = await setup([
  469. // Parent turn (a plain agent): must NOT see the tool.
  470. textResponse('parent answer'),
  471. // Child turn: must see it, with the run's schema.
  472. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
  473. ])
  474. parent.followup(createUserMessage({ content: [{ type: 'text', text: 'hello' }], source: { kind: 'user' } }))
  475. await parent.whenIdle()
  476. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  477. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  478. await run.result
  479. const childRequest = adapter.requests[1]!
  480. expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL)
  481. const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  482. expect(entry.parameters).toEqual(SCHEMA)
  483. await run.dispose()
  484. })
  485. it('two concurrent structured children each see their OWN schema', async () => {
  486. const otherSchema: ObjectJsonSchema = {
  487. type: 'object',
  488. properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } },
  489. required: ['verdict'],
  490. }
  491. const { ctx, parent, adapter } = await setup([
  492. (options: GenerateOptions) => {
  493. // Answer with whatever schema this child was given — proves each
  494. // request carried the right one regardless of scheduling order.
  495. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  496. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  497. ? { verdict: 'real' }
  498. : { answer: 1 }
  499. return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args)
  500. },
  501. (options: GenerateOptions) => {
  502. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  503. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  504. ? { verdict: 'real' }
  505. : { answer: 1 }
  506. return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args)
  507. },
  508. ])
  509. const runA = await ctx.subagents.start('spawn', structuredRequest(parent))
  510. const runB = await ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema }))
  511. const [a, b] = await Promise.all([runA.result, runB.result])
  512. expect(a.structured).toEqual({ answer: 1 })
  513. expect(b.structured).toEqual({ verdict: 'real' })
  514. const schemas = adapter.requests.map(request =>
  515. request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters)
  516. expect(schemas).toContainEqual(SCHEMA)
  517. expect(schemas).toContainEqual(otherSchema)
  518. await runA.dispose()
  519. await runB.dispose()
  520. })
  521. it('places the capture tool and instruction in their canonical orders', async () => {
  522. const { ctx, parent, adapter } = await setup([
  523. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  524. ])
  525. // A global tool sorts lexicographically after structured_output, while a
  526. // global section after the final-output slot follows the capture instruction.
  527. ctx.tools.register(defineContentToolFixture({
  528. name: 'zz_probe',
  529. description: 'probe',
  530. parameters: {},
  531. execute: () => Promise.resolve([{ type: 'text', text: 'x' }]),
  532. }))
  533. ctx.systemPrompt.section({
  534. name: 'after-band',
  535. order: ctx.systemPrompt.getSectionOrder('STRUCTURED_OUTPUT') + 10,
  536. text: 'AFTER-BAND',
  537. })
  538. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  539. await run.result
  540. const request = adapter.requests[0]!
  541. const names = toolNames(request)
  542. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeGreaterThanOrEqual(0)
  543. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeLessThan(names.indexOf('zz_probe'))
  544. const system = requestSystem(request)
  545. const instructionAt = system.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)
  546. expect(instructionAt).toBeGreaterThanOrEqual(0)
  547. expect(system.indexOf('AFTER-BAND')).toBeGreaterThan(instructionAt)
  548. await run.dispose()
  549. })
  550. it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => {
  551. const { parent, adapter } = await setup([textResponse('plain')])
  552. parent.followup(createUserMessage({ content: [{ type: 'text', text: 'q' }], source: { kind: 'user' } }))
  553. await parent.whenIdle()
  554. const request = adapter.requests[0]!
  555. expect(request.tools).toBeUndefined()
  556. await new Promise(resolve => setTimeout(resolve, 0))
  557. })
  558. it('registrations ride the child fiber: disposing the run removes them; a provider reload mid-run cannot', async () => {
  559. const { ctx, parent, disposeProvider } = await setup([
  560. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }),
  561. ])
  562. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  563. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  564. // A backend hot-reload mid-run must not unregister the capture tool out
  565. // from under the live child: the registration rides the CHILD's fiber.
  566. disposeProvider()
  567. const result = await run.result
  568. expect(result.structured).toEqual({ answer: 4 })
  569. const child = ctx.agents.get(run.id)!
  570. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeDefined()
  571. await run.dispose()
  572. // Child disposed ⇒ its scoped registrations are gone.
  573. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeUndefined()
  574. })
  575. })
  576. it('a structured_output call from an agent WITHOUT a structured run is UNKNOWN_TOOL (the tool does not exist for it)', async () => {
  577. const { ctx, parent } = await setup([])
  578. const result = await ctx.tools.execute({
  579. signal: testToolSignal,
  580. callId: 'x' as never,
  581. name: STRUCTURED_OUTPUT_TOOL,
  582. arguments: { answer: 1 },
  583. agent: parent,
  584. })
  585. expect(result.isError).toBe(true)
  586. expect(result.error?.info?.code).toBe('UNKNOWN_TOOL')
  587. })
  588. it('a structured_output call with NO calling agent at all is UNKNOWN_TOOL', async () => {
  589. const { ctx } = await setup([])
  590. const result = await ctx.tools.execute({
  591. signal: testToolSignal,
  592. callId: 'x' as never,
  593. name: STRUCTURED_OUTPUT_TOOL,
  594. arguments: { answer: 1 },
  595. })
  596. expect(result.isError).toBe(true)
  597. expect(result.error?.info?.code).toBe('UNKNOWN_TOOL')
  598. })
  599. it('a failed execution stage is discarded and never promoted by a later call', async () => {
  600. const { ctx, parent } = await setup([
  601. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  602. ])
  603. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  604. // A prepended post-execute listener blocks the first capture without
  605. // delegating. The final-result notification discards that execution's
  606. // stage when it observes the error.
  607. let blocks = 1
  608. ctx.on('tools/post-execute', (exec, _result, next) => {
  609. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  610. blocks -= 1
  611. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  612. }
  613. return next()
  614. }, { prepend: true })
  615. const result = await run.result
  616. const child = ctx.agents.get(run.id)!
  617. // The blocked capture must NOT surface as structured success…
  618. expect(result.stopReason).toBe('error')
  619. expect(result.structured).toBeUndefined()
  620. // …and a LATER invalid call (its own body staged nothing) must not
  621. // resurrect c1's discarded value: drive the pipeline directly.
  622. const invalid = await ctx.tools.execute({
  623. signal: testToolSignal,
  624. callId: 'c2' as never,
  625. name: STRUCTURED_OUTPUT_TOOL,
  626. arguments: { answer: 'not-a-number' },
  627. agent: child,
  628. })
  629. expect(invalid.isError).toBe(true)
  630. // A fresh valid call still captures ITS OWN value.
  631. const valid = await ctx.tools.execute({
  632. signal: testToolSignal,
  633. callId: 'c3' as never,
  634. name: STRUCTURED_OUTPUT_TOOL,
  635. arguments: { answer: 9 },
  636. agent: child,
  637. })
  638. expect(valid.isError).toBeFalsy()
  639. await run.dispose()
  640. })
  641. it('reusing a failed execution\'s call id never promotes its discarded stage', async () => {
  642. const { ctx, parent } = await setup([
  643. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  644. ])
  645. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  646. // Block the first capture after its body stages a value. Its final error
  647. // discards that execution's stage.
  648. let blocks = 1
  649. ctx.on('tools/post-execute', (exec, _result, next) => {
  650. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  651. blocks -= 1
  652. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  653. }
  654. return next()
  655. }, { prepend: true })
  656. await run.result
  657. const child = ctx.agents.get(run.id)!
  658. // A SECOND capture call with the SAME call id whose body never stages
  659. // (invalid args throw before the stage): the discarded value must not ride
  660. // its acceptance.
  661. const reused = await ctx.tools.execute({
  662. signal: testToolSignal,
  663. callId: 'c1' as never,
  664. name: STRUCTURED_OUTPUT_TOOL,
  665. arguments: { answer: 'not-a-number' },
  666. agent: child,
  667. })
  668. expect(reused.isError).toBe(true)
  669. // Nothing was ever committed: a fresh valid call is still required.
  670. const valid = await ctx.tools.execute({
  671. signal: testToolSignal,
  672. callId: 'c1' as never,
  673. name: STRUCTURED_OUTPUT_TOOL,
  674. arguments: { answer: 5 },
  675. agent: child,
  676. })
  677. expect(valid.isError).toBeFalsy()
  678. await run.dispose()
  679. })
  680. it('a pre-execute deny with call-id reuse cannot promote another execution\'s stage', async () => {
  681. const { ctx, parent } = await setup([
  682. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  683. ])
  684. const run = await ctx.subagents.start('spawn', structuredRequest(parent))
  685. // Discard the first capture's stage via a final post-execute block.
  686. let blocks = 1
  687. ctx.on('tools/post-execute', (exec, _result, next) => {
  688. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  689. blocks -= 1
  690. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  691. }
  692. return next()
  693. }, { prepend: true })
  694. await run.result
  695. const child = ctx.agents.get(run.id)!
  696. // A prepended pre-execute deny skips the body, while the denied call still
  697. // reaches the final notification with the same adapter-minted call id.
  698. const offDeny = ctx.on('tools/pre-execute', (exec) => {
  699. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  700. return Promise.resolve({ kind: 'deny' as const, reason: 'outer veto' })
  701. }
  702. return undefined as never
  703. }, { prepend: true })
  704. const denied = await ctx.tools.execute({
  705. signal: testToolSignal,
  706. callId: 'c1' as never,
  707. name: STRUCTURED_OUTPUT_TOOL,
  708. arguments: { answer: 2 },
  709. agent: child,
  710. })
  711. expect(denied.isError).toBe(true)
  712. offDeny()
  713. // The discarded value was never promoted: a fresh valid call is required
  714. // (and succeeds, proving the runtime is not wedged).
  715. const valid = await ctx.tools.execute({
  716. signal: testToolSignal,
  717. callId: 'c1' as never,
  718. name: STRUCTURED_OUTPUT_TOOL,
  719. arguments: { answer: 5 },
  720. agent: child,
  721. })
  722. expect(valid.isError).toBeFalsy()
  723. await run.dispose()
  724. })
  725. })