structured.spec.ts 33 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm'
  4. import SessionStore from '@deepseek-ai/dsh-session'
  5. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  6. import ToolRegistry from '@deepseek-ai/dsh-tools'
  7. import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
  8. import type { ContinuationDecision } from '@deepseek-ai/dsh-agent'
  9. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  10. import * as Invariants from '@deepseek-ai/dsh-invariants'
  11. import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent'
  12. import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
  13. import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
  14. import { startInProcessRun } from '../src/index.ts'
  15. import {
  16. STRUCTURED_OUTPUT_INSTRUCTION,
  17. STRUCTURED_OUTPUT_TOOL,
  18. } from '../src/structured.ts'
  19. type Script = ConstructorParameters<typeof MockAdapter>[0]
  20. const SCHEMA: StructuredOutputSchema = {
  21. type: 'object',
  22. properties: { answer: { type: 'number' }, note: { type: 'string' } },
  23. required: ['answer'],
  24. }
  25. /**
  26. * Real loop + scripted mock model + an INLINE spawn-shaped provider over the
  27. * shared driver. The concrete backend plugins are deliberately NOT loaded —
  28. * they would devDep-cycle this package (spawn/fork already depend on the
  29. * driver), and the runtime under test is the driver's; plugin-level structured
  30. * coverage lives in the spawn/fork specs. The mock model script drives the
  31. * child's structured_output calls.
  32. */
  33. async function setup(script: Script) {
  34. const ctx = new Context()
  35. const adapter = new MockAdapter(script)
  36. await ctx.plugin(LlmService)
  37. await ctx.plugin(SessionStore)
  38. await ctx.plugin(SystemPrompt)
  39. await ctx.plugin(ToolRegistry)
  40. await ctx.plugin(AgentRegistry)
  41. await ctx.plugin(Invariants)
  42. await ctx.plugin(AgentLoop, { agents: [] })
  43. await ctx.plugin(SubagentService)
  44. const disposeProvider = ctx.subagents.registerProvider({
  45. name: 'spawn',
  46. capabilities: { outputSchema: true, depthLimit: true, toolFilter: false, persona: false },
  47. inheritsParentContext: false,
  48. start: (request: SubagentStartRequest) => startInProcessRun(ctx, request, { providerName: 'spawn' }),
  49. })
  50. ctx.llm.registerAdapter(['mock'], adapter)
  51. const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' })
  52. return { ctx, parent, adapter, disposeProvider }
  53. }
  54. function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial<SubagentStartRequest>): SubagentStartRequest {
  55. return { prompt: [{ type: 'text', text: 'produce the answer' }], parent, outputSchema: SCHEMA, ...extra }
  56. }
  57. /** The tool names of one recorded model request. */
  58. function toolNames(request: GenerateOptions): string[] {
  59. return (request.tools ?? []).map(tool => tool.name)
  60. }
  61. describe('in-process structured output', () => {
  62. it('captures a valid structured_output call and surfaces result.structured', async () => {
  63. const { ctx, parent } = await setup([
  64. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }),
  65. ])
  66. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  67. const result = await run.result
  68. expect(result.stopReason).toBe('completed')
  69. expect(result.structured).toEqual({ answer: 42, note: 'done' })
  70. await run.dispose()
  71. })
  72. it('stops the turn after a successful capture — no extra model step is spent', async () => {
  73. const { ctx, parent, adapter } = await setup([
  74. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  75. textResponse('MUST NOT BE CONSUMED'),
  76. ])
  77. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  78. await run.result
  79. // Default continuation would run a second step after the tool call; the
  80. // structured runtime's turn-continuation veto stops the turn instead.
  81. expect(adapter.requests.length).toBe(1)
  82. await run.dispose()
  83. })
  84. it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => {
  85. // One model response carrying structured_output FIRST and a side-effecting
  86. // call after it: the continuation veto only fires at step end, so without
  87. // the pre-execute deny the trailing call would still run after the final
  88. // answer was accepted.
  89. const response = [
  90. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  91. { type: 'block-start', index: 1, blockType: 'tool-call' },
  92. { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
  93. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  94. { type: 'finish', reason: { kind: 'tool-calls' } },
  95. ] as Script[number]
  96. const { ctx, parent } = await setup([response])
  97. let sideEffectRan = false
  98. ctx.tools.register({
  99. name: 'side_effect',
  100. description: 'probe',
  101. parameters: { type: 'object', properties: {} },
  102. execute(): Promise<ContentBlock[]> {
  103. sideEffectRan = true
  104. return Promise.resolve([{ type: 'text', text: 'ran' }])
  105. },
  106. })
  107. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  108. const result = await run.result
  109. expect(result.stopReason).toBe('completed')
  110. expect(result.structured).toEqual({ answer: 5 })
  111. // The deny skipped dispatch entirely: the probe body never ran.
  112. expect(sideEffectRan).toBe(false)
  113. await run.dispose()
  114. })
  115. it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => {
  116. const response = [
  117. { type: 'block-start', index: 0, blockType: 'tool-call' },
  118. { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'side_effect', arguments: '{}' } },
  119. ...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk =>
  120. 'index' in chunk ? { ...chunk, index: 1 } : chunk),
  121. ] as Script[number]
  122. const { ctx, parent } = await setup([response])
  123. let sideEffectRan = false
  124. ctx.tools.register({
  125. name: 'side_effect',
  126. description: 'probe',
  127. parameters: { type: 'object', properties: {} },
  128. execute(): Promise<ContentBlock[]> {
  129. sideEffectRan = true
  130. return Promise.resolve([{ type: 'text', text: 'ran' }])
  131. },
  132. })
  133. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  134. const result = await run.result
  135. // The call ran BEFORE captured was set: the deny gate only guards the
  136. // window after the terminal answer landed.
  137. expect(sideEffectRan).toBe(true)
  138. expect(result.structured).toEqual({ answer: 6 })
  139. await run.dispose()
  140. })
  141. it('snapshots the schema at start(): caller mutation after start cannot drift enforcement', async () => {
  142. const mutable: StructuredOutputSchema = {
  143. type: 'object',
  144. properties: { answer: { type: 'number' } },
  145. required: ['answer'],
  146. additionalProperties: false,
  147. }
  148. const pristine = structuredClone(mutable)
  149. const { ctx, parent, adapter } = await setup([
  150. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }),
  151. ])
  152. const run = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: mutable }))
  153. // Mutate the caller's object AFTER start() returned but before the child's
  154. // first request assembles: with a live reference this would reach both the
  155. // model-visible parameters and validateStructuredValue.
  156. ;(mutable.properties as Record<string, unknown>).answer = { type: 'string' }
  157. const result = await run.result
  158. expect(result.structured).toEqual({ answer: 3 })
  159. // The child's request carried the PRISTINE schema, not the mutated one.
  160. const childRequest = adapter.requests.at(-1)
  161. const captureTool = (childRequest?.tools ?? []).find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
  162. expect(captureTool?.parameters).toEqual(pristine)
  163. await run.dispose()
  164. })
  165. it('the captured-turn veto is prepend: an EARLIER force-continue listener cannot short-circuit it', async () => {
  166. // A goal-style listener registered BEFORE the child exists, returning a
  167. // forced continue WITHOUT calling next(). Without prepend on the scoped
  168. // veto, this would decide the turn first and buy a wasted model step —
  169. // the one-response script would then throw on the second request.
  170. const { ctx, parent, adapter } = await setup([
  171. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  172. ])
  173. ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'continue' }))
  174. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  175. const result = await run.result
  176. expect(result.structured).toEqual({ answer: 7 })
  177. expect(result.stopReason).toBe('completed')
  178. expect(adapter.requests).toHaveLength(1)
  179. await run.dispose()
  180. })
  181. it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => {
  182. const { ctx, parent } = await setup([
  183. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }),
  184. toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  185. ])
  186. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  187. const result = await run.result
  188. expect(result.structured).toEqual({ answer: 7 })
  189. expect(result.stopReason).toBe('completed')
  190. // The child's log carries the isError tool/result for the invalid call.
  191. const child = ctx.agents.get(run.id)!
  192. const results = child.session.events.filter(e => e.type === 'tool/result')
  193. expect(results.length).toBe(2)
  194. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  195. await run.dispose()
  196. })
  197. it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => {
  198. const { ctx, parent, adapter } = await setup([
  199. textResponse('here is my answer in prose'),
  200. textResponse('MUST NOT BE CONSUMED'),
  201. ])
  202. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  203. const result = await run.result
  204. expect(result.stopReason).toBe('error')
  205. expect(result.structured).toBeUndefined()
  206. // Exactly one model request and one user message: no nudge turn exists.
  207. expect(adapter.requests.length).toBe(1)
  208. const child = ctx.agents.get(run.id)!
  209. expect(child.session.events.filter(e => e.type === 'user/message').length).toBe(1)
  210. await run.dispose()
  211. })
  212. it('an errored child keeps its honest error result (no capture expected)', async () => {
  213. // Script exhaustion on the first call → the child turn errors.
  214. const { ctx, parent, adapter } = await setup([])
  215. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  216. const result = await run.result
  217. expect(result.stopReason).toBe('error')
  218. expect(adapter.requests.length).toBe(1)
  219. await run.dispose()
  220. })
  221. it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => {
  222. const { ctx, parent } = await setup([textResponse('prose, no capture')])
  223. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  224. const child = ctx.agents.get(run.id)!
  225. // Cancel synchronously inside the turn's end recording: the cancel
  226. // contract outranks the schema shortfall, so the result maps to aborted.
  227. ctx.on('session/event', (session, event) => {
  228. if (session === child.session && event.type === 'turn/end') run.cancel('cancelled at turn end')
  229. })
  230. const result = await run.result
  231. expect(result.stopReason).toBe('aborted')
  232. await run.dispose()
  233. })
  234. it('rejects a schema outside the subset loud, before any child exists', async () => {
  235. const { ctx, parent } = await setup([])
  236. expect(() => ctx.subagents.start('spawn', structuredRequest(parent, {
  237. outputSchema: { type: 'object', oneOf: [] } as unknown as StructuredOutputSchema,
  238. }))).toThrow(/unsupported output schema/)
  239. expect(ctx.agents.get(AgentId('parent'))).toBeDefined()
  240. })
  241. it('a schema carrying non-JSON values fails as OutputSchemaError, never as a raw clone error', async () => {
  242. const { ctx, parent } = await setup([])
  243. // Assertion runs BEFORE the defensive structuredClone: a function-valued
  244. // annotation must surface as the subset violation it is, not escape as
  245. // structuredClone's DataCloneError.
  246. expect(() => ctx.subagents.start('spawn', structuredRequest(parent, {
  247. outputSchema: { type: 'object', default: () => {} } as unknown as StructuredOutputSchema,
  248. }))).toThrow(/unsupported output schema.*annotation must be JSON data/)
  249. })
  250. it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => {
  251. const { ctx, parent, adapter } = await setup([
  252. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  253. textResponse('continues after the blocked capture'),
  254. ])
  255. // A PostToolUse-style hook, registered AFTER the runtime (so the runtime's
  256. // prepend commit listener stays outermost and composes this verdict).
  257. ctx.on('tools/post-execute', (exec, _result, next) => {
  258. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  259. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] })
  260. }
  261. return next()
  262. })
  263. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  264. const result = await run.result
  265. // No capture was committed: the run reports the schema shortfall...
  266. expect(result.structured).toBeUndefined()
  267. expect(result.stopReason).toBe('error')
  268. // ...the logged tool result is the blocked isError with the feedback...
  269. const child = ctx.agents.get(run.id)!
  270. const results = child.session.events.filter(e => e.type === 'tool/result')
  271. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  272. expect(JSON.stringify((results[0]!.data as { content: unknown }).content)).toContain('capture rejected by hook')
  273. // ...and the turn CONTINUED past the blocked call (no captured veto):
  274. // the model got to react to the failure with a second step.
  275. expect(adapter.requests.length).toBe(2)
  276. await run.dispose()
  277. })
  278. it('a post-execute accept-with-replacement still commits the capture', async () => {
  279. const { ctx, parent } = await setup([
  280. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  281. ])
  282. ctx.on('tools/post-execute', (exec, _result, next) => {
  283. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  284. return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] })
  285. }
  286. return next()
  287. })
  288. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  289. const result = await run.result
  290. expect(result.stopReason).toBe('completed')
  291. expect(result.structured).toEqual({ answer: 8 })
  292. await run.dispose()
  293. })
  294. it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => {
  295. const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })])
  296. // A context-wide section stands in for the deployment persona: the
  297. // instruction must APPEND to whatever the prompt pipeline assembled, not
  298. // replace it (AgentOptions has no prompt field — the instruction is
  299. // per-request wire state added by the final-request listener).
  300. ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' })
  301. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  302. await run.result
  303. const childRequest = adapter.requests.at(-1)!
  304. expect(childRequest.system).toContain('You are a counter.')
  305. expect(childRequest.system!.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  306. expect(childRequest.system!.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0)
  307. await run.dispose()
  308. })
  309. it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => {
  310. const { ctx, parent, adapter } = await setup([
  311. textResponse('parent answer'),
  312. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  313. ])
  314. parent.send([{ type: 'text', text: 'hello' }])
  315. await parent.whenIdle()
  316. expect(adapter.requests[0]!.system ?? '').not.toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  317. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  318. await run.result
  319. // The loop always assembles a base prompt (the harness identity section),
  320. // so the instruction APPENDS — never replaces.
  321. const childSystem = adapter.requests.at(-1)!.system!
  322. expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  323. expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length)
  324. await run.dispose()
  325. })
  326. describe('scoped registration (each child owns its capture tool)', () => {
  327. it('a plain agent never sees the tool: nothing is registered globally at all', async () => {
  328. const { ctx, parent, adapter } = await setup([textResponse('parent answer')])
  329. parent.send([{ type: 'text', text: 'hello' }])
  330. await parent.whenIdle()
  331. // Scoped registration: the global view has no capture tool, ever.
  332. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  333. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  334. })
  335. it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => {
  336. const { ctx, parent, adapter } = await setup([
  337. // Parent turn (a plain agent): must NOT see the tool.
  338. textResponse('parent answer'),
  339. // Child turn: must see it, with the run's schema.
  340. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
  341. ])
  342. parent.send([{ type: 'text', text: 'hello' }])
  343. await parent.whenIdle()
  344. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  345. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  346. await run.result
  347. const childRequest = adapter.requests[1]!
  348. expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL)
  349. const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  350. expect(entry.parameters).toEqual(SCHEMA)
  351. await run.dispose()
  352. })
  353. it('two concurrent structured children each see their OWN schema', async () => {
  354. const otherSchema: StructuredOutputSchema = {
  355. type: 'object',
  356. properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } },
  357. required: ['verdict'],
  358. }
  359. const { ctx, parent, adapter } = await setup([
  360. (options: GenerateOptions) => {
  361. // Answer with whatever schema this child was given — proves each
  362. // request carried the right one regardless of scheduling order.
  363. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  364. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  365. ? { verdict: 'real' }
  366. : { answer: 1 }
  367. return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args)
  368. },
  369. (options: GenerateOptions) => {
  370. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  371. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  372. ? { verdict: 'real' }
  373. : { answer: 1 }
  374. return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args)
  375. },
  376. ])
  377. const runA = ctx.subagents.start('spawn', structuredRequest(parent))
  378. const runB = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema }))
  379. const [a, b] = await Promise.all([runA.result, runB.result])
  380. expect(a.structured).toEqual({ answer: 1 })
  381. expect(b.structured).toEqual({ verdict: 'real' })
  382. const schemas = adapter.requests.map(request =>
  383. request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters)
  384. expect(schemas).toContainEqual(SCHEMA)
  385. expect(schemas).toContainEqual(otherSchema)
  386. await runA.dispose()
  387. await runB.dispose()
  388. })
  389. it('the re-assert REPLACES a conflicting injected schema, not merely ensures presence', async () => {
  390. const { ctx, parent, adapter } = await setup([
  391. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }),
  392. ])
  393. // A global listener that INJECTS a wrong-schema structured_output entry:
  394. // the child's re-assert must replace it with the run's own schema.
  395. ctx.on('system-prompt/assemble', async (_assembly, _context, next) => {
  396. const replaced = await next()
  397. return {
  398. sections: replaced.sections,
  399. tools: [
  400. ...replaced.tools.filter(tool => tool.name !== STRUCTURED_OUTPUT_TOOL),
  401. { name: STRUCTURED_OUTPUT_TOOL, description: 'wrong', parameters: { type: 'object', properties: { bogus: { type: 'string' } } } },
  402. ],
  403. variables: { ...replaced.variables },
  404. }
  405. })
  406. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  407. const result = await run.result
  408. expect(result.structured).toEqual({ answer: 5 })
  409. const entries = adapter.requests[0]!.tools!.filter(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
  410. expect(entries).toHaveLength(1)
  411. expect(entries[0]!.parameters).toEqual(SCHEMA)
  412. await run.dispose()
  413. })
  414. it('the re-assert wins against a downstream listener that REPLACES the assembly object', async () => {
  415. const { ctx, parent, adapter } = await setup([
  416. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }),
  417. ])
  418. // A global (every-assembly) listener that returns a brand-new assembly
  419. // WITHOUT the capture tool or instruction — the composition caveat that
  420. // erases cooperative mutations. The child's prepend re-assert runs
  421. // OUTERMOST and restores both.
  422. ctx.on('system-prompt/assemble', async (_assembly, _context, next) => {
  423. const replaced = await next()
  424. return {
  425. sections: replaced.sections.filter(section => section.name !== `tool:${STRUCTURED_OUTPUT_TOOL}`),
  426. tools: replaced.tools.filter(tool => tool.name !== STRUCTURED_OUTPUT_TOOL),
  427. variables: { ...replaced.variables },
  428. }
  429. })
  430. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  431. const result = await run.result
  432. expect(result.structured).toEqual({ answer: 5 })
  433. const entry = adapter.requests[0]!.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
  434. expect(entry).toBeDefined()
  435. expect(entry!.parameters).toEqual(SCHEMA)
  436. const system = adapter.requests[0]!.system ?? ''
  437. expect(system).toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  438. await run.dispose()
  439. })
  440. it('the re-assert preserves the untampered assembly: tool position and section band are the registry\'s own', async () => {
  441. const { ctx, parent, adapter } = await setup([
  442. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  443. ])
  444. // A global tool sorting lexicographically AFTER structured_output and a
  445. // global section ABOVE the 190 band: the re-assert must leave both
  446. // exactly where the registry's ordering put them (no move-to-end).
  447. ctx.tools.register({
  448. name: 'zz_probe',
  449. description: 'probe',
  450. parameters: { type: 'object', properties: {} },
  451. execute: () => Promise.resolve([{ type: 'text', text: 'x' }]),
  452. })
  453. ctx.systemPrompt.section({ name: 'after-band', order: 200, text: 'AFTER-BAND' })
  454. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  455. await run.result
  456. const request = adapter.requests[0]!
  457. const names = toolNames(request)
  458. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeGreaterThanOrEqual(0)
  459. expect(names.indexOf(STRUCTURED_OUTPUT_TOOL)).toBeLessThan(names.indexOf('zz_probe'))
  460. const system = request.system ?? ''
  461. const instructionAt = system.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)
  462. expect(instructionAt).toBeGreaterThanOrEqual(0)
  463. expect(system.indexOf('AFTER-BAND')).toBeGreaterThan(instructionAt)
  464. await run.dispose()
  465. })
  466. it('a stripped instruction re-inserts at its band; an added duplicate entry collapses to one', async () => {
  467. const { ctx, parent, adapter } = await setup([
  468. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }),
  469. ])
  470. ctx.systemPrompt.section({ name: 'after-band', order: 200, text: 'AFTER-BAND' })
  471. // Strip the instruction section entirely AND add a wrong-schema
  472. // duplicate tool entry ALONGSIDE the registry's own: the re-assert must
  473. // restore the section INTO its band (before the order-200 section, not
  474. // appended after it) and collapse the tools to exactly one entry
  475. // carrying the run's schema.
  476. ctx.on('system-prompt/assemble', async (_assembly, _context, next) => {
  477. const replaced = await next()
  478. return {
  479. sections: replaced.sections.filter(section => section.name !== `tool:${STRUCTURED_OUTPUT_TOOL}`),
  480. tools: [
  481. ...replaced.tools,
  482. { name: STRUCTURED_OUTPUT_TOOL, description: 'wrong', parameters: { type: 'object', properties: { bogus: { type: 'string' } } } },
  483. ],
  484. variables: { ...replaced.variables },
  485. }
  486. })
  487. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  488. const result = await run.result
  489. expect(result.structured).toEqual({ answer: 3 })
  490. const request = adapter.requests[0]!
  491. const entries = request.tools!.filter(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
  492. expect(entries).toHaveLength(1)
  493. expect(entries[0]!.parameters).toEqual(SCHEMA)
  494. const system = request.system ?? ''
  495. const instructionAt = system.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)
  496. expect(instructionAt).toBeGreaterThanOrEqual(0)
  497. expect(system.indexOf('AFTER-BAND')).toBeGreaterThan(instructionAt)
  498. await run.dispose()
  499. })
  500. it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => {
  501. const { parent, adapter } = await setup([textResponse('plain')])
  502. parent.send([{ type: 'text', text: 'q' }])
  503. await parent.whenIdle()
  504. const request = adapter.requests[0]!
  505. expect(request.tools).toBeUndefined()
  506. await new Promise(resolve => setTimeout(resolve, 0))
  507. })
  508. it('registrations ride the child fiber: disposing the run removes them; a provider reload mid-run cannot', async () => {
  509. const { ctx, parent, disposeProvider } = await setup([
  510. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }),
  511. ])
  512. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  513. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  514. // A backend hot-reload mid-run must not unregister the capture tool out
  515. // from under the live child: the registration rides the CHILD's fiber.
  516. await disposeProvider()
  517. const result = await run.result
  518. expect(result.structured).toEqual({ answer: 4 })
  519. const child = ctx.agents.get(run.id)!
  520. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeDefined()
  521. await run.dispose()
  522. // Child disposed ⇒ its scoped registrations are gone.
  523. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL, child)).toBeUndefined()
  524. })
  525. })
  526. it('a structured_output call from an agent WITHOUT a structured run is UNKNOWN_TOOL (the tool does not exist for it)', async () => {
  527. const { ctx, parent } = await setup([])
  528. const result = await ctx.tools.execute({
  529. callId: 'x' as never,
  530. name: STRUCTURED_OUTPUT_TOOL,
  531. arguments: { answer: 1 },
  532. agent: parent,
  533. })
  534. expect(result.isError).toBe(true)
  535. expect(result.error?.code).toBe('UNKNOWN_TOOL')
  536. })
  537. it('a structured_output call with NO calling agent at all is UNKNOWN_TOOL', async () => {
  538. const { ctx } = await setup([])
  539. const result = await ctx.tools.execute({
  540. callId: 'x' as never,
  541. name: STRUCTURED_OUTPUT_TOOL,
  542. arguments: { answer: 1 },
  543. })
  544. expect(result.isError).toBe(true)
  545. expect(result.error?.code).toBe('UNKNOWN_TOOL')
  546. })
  547. it('a stale stage from a short-circuited chain is never promoted by a later call (execution-keyed commit)', async () => {
  548. const { ctx, parent } = await setup([
  549. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  550. ])
  551. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  552. const child = ctx.agents.get(run.id)!
  553. // An OUTER post-execute listener (registered after attach, prepend ⇒
  554. // outermost) that BLOCKS the first capture WITHOUT delegating: the commit
  555. // listener never runs for c1, so its staged value would linger.
  556. let blocks = 1
  557. ctx.on('tools/post-execute', (exec, _result, next) => {
  558. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  559. blocks -= 1
  560. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  561. }
  562. return next()
  563. }, { prepend: true })
  564. const result = await run.result
  565. // The blocked capture must NOT surface as structured success…
  566. expect(result.stopReason).toBe('error')
  567. expect(result.structured).toBeUndefined()
  568. // …and a LATER invalid call (its own body staged nothing) must not
  569. // resurrect c1's orphaned value: drive the pipeline directly.
  570. const invalid = await ctx.tools.execute({
  571. callId: 'c2' as never,
  572. name: STRUCTURED_OUTPUT_TOOL,
  573. arguments: { answer: 'not-a-number' },
  574. agent: child,
  575. })
  576. expect(invalid.isError).toBe(true)
  577. // A fresh valid call still captures ITS OWN value.
  578. const valid = await ctx.tools.execute({
  579. callId: 'c3' as never,
  580. name: STRUCTURED_OUTPUT_TOOL,
  581. arguments: { answer: 9 },
  582. agent: child,
  583. })
  584. expect(valid.isError).toBeFalsy()
  585. await run.dispose()
  586. })
  587. it('a later capture call REUSING a stale stage\'s call id never promotes it (unconditional commit safety)', async () => {
  588. const { ctx, parent } = await setup([
  589. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  590. ])
  591. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  592. const child = ctx.agents.get(run.id)!
  593. // Orphan a stage: an outer short-circuiting post-execute BLOCK on the
  594. // first capture (its chain never reaches the commit listener).
  595. let blocks = 1
  596. ctx.on('tools/post-execute', (exec, _result, next) => {
  597. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  598. blocks -= 1
  599. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  600. }
  601. return next()
  602. }, { prepend: true })
  603. await run.result
  604. // A SECOND capture call with the SAME call id whose body never stages
  605. // (invalid args throw before the stage): the stale value must not ride
  606. // its acceptance.
  607. const reused = await ctx.tools.execute({
  608. callId: 'c1' as never,
  609. name: STRUCTURED_OUTPUT_TOOL,
  610. arguments: { answer: 'not-a-number' },
  611. agent: child,
  612. })
  613. expect(reused.isError).toBe(true)
  614. // Nothing was ever committed: a fresh valid call is still required.
  615. const valid = await ctx.tools.execute({
  616. callId: 'c1' as never,
  617. name: STRUCTURED_OUTPUT_TOOL,
  618. arguments: { answer: 5 },
  619. agent: child,
  620. })
  621. expect(valid.isError).toBeFalsy()
  622. await run.dispose()
  623. })
  624. it('an outer pre-execute deny with call-id reuse cannot promote an orphaned stage either', async () => {
  625. const { ctx, parent } = await setup([
  626. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  627. ])
  628. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  629. const child = ctx.agents.get(run.id)!
  630. // Orphan a stage via an outer post-execute BLOCK on the first capture.
  631. let blocks = 1
  632. ctx.on('tools/post-execute', (exec, _result, next) => {
  633. if (exec.name === STRUCTURED_OUTPUT_TOOL && blocks > 0) {
  634. blocks -= 1
  635. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'rejected' }] })
  636. }
  637. return next()
  638. }, { prepend: true })
  639. await run.result
  640. // An OUTERMOST prepend pre-execute deny: the structured runtime's own
  641. // pre-execute never runs for this call, and the denied call still goes
  642. // through post-execute — with the SAME call id as the orphaned stage.
  643. const offDeny = ctx.on('tools/pre-execute', (exec) => {
  644. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  645. return Promise.resolve({ kind: 'deny' as const, reason: 'outer veto' })
  646. }
  647. return undefined as never
  648. }, { prepend: true })
  649. const denied = await ctx.tools.execute({
  650. callId: 'c1' as never,
  651. name: STRUCTURED_OUTPUT_TOOL,
  652. arguments: { answer: 2 },
  653. agent: child,
  654. })
  655. expect(denied.isError).toBe(true)
  656. offDeny()
  657. // The orphan was never promoted: a fresh valid call is still required
  658. // (and succeeds, proving the runtime is not wedged).
  659. const valid = await ctx.tools.execute({
  660. callId: 'c1' as never,
  661. name: STRUCTURED_OUTPUT_TOOL,
  662. arguments: { answer: 5 },
  663. agent: child,
  664. })
  665. expect(valid.isError).toBeFalsy()
  666. await run.dispose()
  667. })
  668. })