structured.spec.ts 29 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { CallId, type ContentBlock, type GenerateOptions } from '@deepseek-ai/dsh-llm'
  4. import SessionStore from '@deepseek-ai/dsh-session'
  5. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  6. import ToolRegistry from '@deepseek-ai/dsh-tools'
  7. import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
  8. import type { Agent, ContinuationDecision } from '@deepseek-ai/dsh-agent'
  9. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  10. import * as Invariants from '@deepseek-ai/dsh-invariants'
  11. import SubagentService, { type SubagentStartRequest } from '@deepseek-ai/dsh-subagent'
  12. import type { StructuredOutputSchema } from '@deepseek-ai/dsh-tools'
  13. import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
  14. import { startInProcessRun } from '../src/index.ts'
  15. import {
  16. acquireStructuredRuntime,
  17. STRUCTURED_OUTPUT_INSTRUCTION,
  18. STRUCTURED_OUTPUT_TOOL,
  19. } from '../src/structured.ts'
  20. type Script = ConstructorParameters<typeof MockAdapter>[0]
  21. const SCHEMA: StructuredOutputSchema = {
  22. type: 'object',
  23. properties: { answer: { type: 'number' }, note: { type: 'string' } },
  24. required: ['answer'],
  25. }
  26. /**
  27. * Real loop + scripted mock model + an INLINE spawn-shaped provider over the
  28. * shared driver. The concrete backend plugins are deliberately NOT loaded —
  29. * they would devDep-cycle this package (spawn/fork already depend on the
  30. * driver), and the runtime under test is the driver's; plugin-level structured
  31. * coverage lives in the spawn/fork specs. The mock model script drives the
  32. * child's structured_output calls.
  33. */
  34. async function setup(script: Script) {
  35. const ctx = new Context()
  36. const adapter = new MockAdapter(script)
  37. await ctx.plugin(LlmService)
  38. await ctx.plugin(SessionStore)
  39. await ctx.plugin(SystemPrompt)
  40. await ctx.plugin(ToolRegistry)
  41. await ctx.plugin(AgentRegistry)
  42. await ctx.plugin(Invariants)
  43. await ctx.plugin(AgentLoop, { agents: [] })
  44. await ctx.plugin(SubagentService)
  45. const disposeProvider = ctx.subagents.registerProvider({
  46. name: 'spawn',
  47. capabilities: { outputSchema: true, depthLimit: true, toolFilter: false },
  48. inheritsParentContext: false,
  49. start: (request: SubagentStartRequest) => startInProcessRun(ctx, request, { providerName: 'spawn' }),
  50. })
  51. ctx.llm.registerAdapter(['mock'], adapter)
  52. const parent = ctx.agentLoop.create(AgentId('parent'), { model: 'mock' })
  53. return { ctx, parent, adapter, disposeProvider }
  54. }
  55. function structuredRequest(parent: SubagentStartRequest['parent'], extra?: Partial<SubagentStartRequest>): SubagentStartRequest {
  56. return { prompt: [{ type: 'text', text: 'produce the answer' }], parent, outputSchema: SCHEMA, ...extra }
  57. }
  58. /** The tool names of one recorded model request. */
  59. function toolNames(request: GenerateOptions): string[] {
  60. return (request.tools ?? []).map(tool => tool.name)
  61. }
  62. describe('in-process structured output', () => {
  63. it('captures a valid structured_output call and surfaces result.structured', async () => {
  64. const { ctx, parent } = await setup([
  65. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42, note: 'done' }),
  66. ])
  67. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  68. const result = await run.result
  69. expect(result.stopReason).toBe('completed')
  70. expect(result.structured).toEqual({ answer: 42, note: 'done' })
  71. await run.dispose()
  72. })
  73. it('stops the turn after a successful capture — no extra model step is spent', async () => {
  74. const { ctx, parent, adapter } = await setup([
  75. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  76. textResponse('MUST NOT BE CONSUMED'),
  77. ])
  78. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  79. await run.result
  80. // Default continuation would run a second step after the tool call; the
  81. // structured runtime's turn-continuation veto stops the turn instead.
  82. expect(adapter.requests.length).toBe(1)
  83. await run.dispose()
  84. })
  85. it('denies tool calls that FOLLOW the capture in the same response — terminal means terminal', async () => {
  86. // One model response carrying structured_output FIRST and a side-effecting
  87. // call after it: the continuation veto only fires at step end, so without
  88. // the pre-execute deny the trailing call would still run after the final
  89. // answer was accepted.
  90. const response = [
  91. ...toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }).slice(0, -2),
  92. { type: 'block-start', index: 1, blockType: 'tool-call' },
  93. { type: 'block-end', index: 1, block: { type: 'tool-call', id: CallId('c2'), name: 'side_effect', arguments: '{}' } },
  94. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  95. { type: 'finish', reason: { kind: 'tool-calls' } },
  96. ] as Script[number]
  97. const { ctx, parent } = await setup([response])
  98. let sideEffectRan = false
  99. ctx.tools.register({
  100. name: 'side_effect',
  101. description: 'probe',
  102. parameters: { type: 'object', properties: {} },
  103. execute(): Promise<ContentBlock[]> {
  104. sideEffectRan = true
  105. return Promise.resolve([{ type: 'text', text: 'ran' }])
  106. },
  107. })
  108. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  109. const result = await run.result
  110. expect(result.stopReason).toBe('completed')
  111. expect(result.structured).toEqual({ answer: 5 })
  112. // The deny skipped dispatch entirely: the probe body never ran.
  113. expect(sideEffectRan).toBe(false)
  114. await run.dispose()
  115. })
  116. it('leaves tool calls that PRECEDE the capture in the same response untouched', async () => {
  117. const response = [
  118. { type: 'block-start', index: 0, blockType: 'tool-call' },
  119. { type: 'block-end', index: 0, block: { type: 'tool-call', id: CallId('c1'), name: 'side_effect', arguments: '{}' } },
  120. ...toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 6 }).map(chunk =>
  121. 'index' in chunk ? { ...chunk, index: 1 } : chunk),
  122. ] as Script[number]
  123. const { ctx, parent } = await setup([response])
  124. let sideEffectRan = false
  125. ctx.tools.register({
  126. name: 'side_effect',
  127. description: 'probe',
  128. parameters: { type: 'object', properties: {} },
  129. execute(): Promise<ContentBlock[]> {
  130. sideEffectRan = true
  131. return Promise.resolve([{ type: 'text', text: 'ran' }])
  132. },
  133. })
  134. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  135. const result = await run.result
  136. // The call ran BEFORE captured was set: the deny gate only guards the
  137. // window after the terminal answer landed.
  138. expect(sideEffectRan).toBe(true)
  139. expect(result.structured).toEqual({ answer: 6 })
  140. await run.dispose()
  141. })
  142. it('snapshots the schema at start(): caller mutation after start cannot drift enforcement', async () => {
  143. const mutable: StructuredOutputSchema = {
  144. type: 'object',
  145. properties: { answer: { type: 'number' } },
  146. required: ['answer'],
  147. additionalProperties: false,
  148. }
  149. const pristine = structuredClone(mutable)
  150. const { ctx, parent, adapter } = await setup([
  151. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 3 }),
  152. ])
  153. const run = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: mutable }))
  154. // Mutate the caller's object AFTER start() returned but before the child's
  155. // first request assembles: with a live reference this would reach both the
  156. // model-visible parameters and validateStructuredValue.
  157. ;(mutable.properties as Record<string, unknown>).answer = { type: 'string' }
  158. const result = await run.result
  159. expect(result.structured).toEqual({ answer: 3 })
  160. // The child's request carried the PRISTINE schema, not the mutated one.
  161. const childRequest = adapter.requests.at(-1)
  162. const captureTool = (childRequest?.tools ?? []).find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
  163. expect(captureTool?.parameters).toEqual(pristine)
  164. await run.dispose()
  165. })
  166. it('the captured-turn veto is prepend: an EARLIER force-continue listener cannot short-circuit it', async () => {
  167. const ctx = new Context()
  168. await ctx.plugin(SystemPrompt)
  169. await ctx.plugin(ToolRegistry)
  170. // Registered BEFORE the structured runtime exists — without prepend, this
  171. // goal-style listener would decide the turn first (returning WITHOUT
  172. // calling next()) and the veto would never run.
  173. ctx.on('agent/turn-continuation', () => Promise.resolve<ContinuationDecision>({ action: 'continue' }))
  174. const acquisition = acquireStructuredRuntime(ctx)
  175. const agent = { id: AgentId('structured-child') } as unknown as Agent
  176. acquisition.attach(agent, SCHEMA)
  177. const captured = await ctx.tools.execute({
  178. callId: 'call-1' as never,
  179. name: STRUCTURED_OUTPUT_TOOL,
  180. arguments: { answer: 1 },
  181. agent,
  182. })
  183. expect(captured.isError).toBeFalsy()
  184. const decision = await ctx.waterfall(
  185. 'agent/turn-continuation', agent, 1,
  186. { action: 'continue' },
  187. () => Promise.resolve<ContinuationDecision>({ action: 'continue' }),
  188. )
  189. expect(decision).toEqual({ action: 'stop' })
  190. acquisition.detach(agent)
  191. acquisition.release()
  192. })
  193. it('an invalid call gets an INVALID_ARGS isError result and the model retries in-turn', async () => {
  194. const { ctx, parent } = await setup([
  195. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 'not-a-number' }),
  196. toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  197. ])
  198. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  199. const result = await run.result
  200. expect(result.structured).toEqual({ answer: 7 })
  201. expect(result.stopReason).toBe('completed')
  202. // The child's log carries the isError tool/result for the invalid call.
  203. const child = ctx.agents.get(run.id)!
  204. const results = child.session.events.filter(e => e.type === 'tool/result')
  205. expect(results.length).toBe(2)
  206. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  207. await run.dispose()
  208. })
  209. it('a clean finish without a capture is an immediate error to the parent — deliberately NO re-prompt', async () => {
  210. const { ctx, parent, adapter } = await setup([
  211. textResponse('here is my answer in prose'),
  212. textResponse('MUST NOT BE CONSUMED'),
  213. ])
  214. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  215. const result = await run.result
  216. expect(result.stopReason).toBe('error')
  217. expect(result.structured).toBeUndefined()
  218. // Exactly one model request and one user message: no nudge turn exists.
  219. expect(adapter.requests.length).toBe(1)
  220. const child = ctx.agents.get(run.id)!
  221. expect(child.session.events.filter(e => e.type === 'user/message').length).toBe(1)
  222. await run.dispose()
  223. })
  224. it('an errored child keeps its honest error result (no capture expected)', async () => {
  225. // Script exhaustion on the first call → the child turn errors.
  226. const { ctx, parent, adapter } = await setup([])
  227. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  228. const result = await run.result
  229. expect(result.stopReason).toBe('error')
  230. expect(adapter.requests.length).toBe(1)
  231. await run.dispose()
  232. })
  233. it('a cancel landing after a clean capture-less turn settles aborted, not error', async () => {
  234. const { ctx, parent } = await setup([textResponse('prose, no capture')])
  235. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  236. const child = ctx.agents.get(run.id)!
  237. // Cancel synchronously inside the turn's end recording: the cancel
  238. // contract outranks the schema shortfall, so the result maps to aborted.
  239. ctx.on('session/event', (session, event) => {
  240. if (session === child.session && event.type === 'turn/end') run.cancel('cancelled at turn end')
  241. })
  242. const result = await run.result
  243. expect(result.stopReason).toBe('aborted')
  244. await run.dispose()
  245. })
  246. it('rejects a schema outside the subset loud, before any child exists', async () => {
  247. const { ctx, parent } = await setup([])
  248. expect(() => ctx.subagents.start('spawn', structuredRequest(parent, {
  249. outputSchema: { type: 'object', oneOf: [] } as unknown as StructuredOutputSchema,
  250. }))).toThrow(/unsupported output schema/)
  251. expect(ctx.agents.get(AgentId('parent'))).toBeDefined()
  252. })
  253. it('a schema carrying non-JSON values fails as OutputSchemaError, never as a raw clone error', async () => {
  254. const { ctx, parent } = await setup([])
  255. // Assertion runs BEFORE the defensive structuredClone: a function-valued
  256. // annotation must surface as the subset violation it is, not escape as
  257. // structuredClone's DataCloneError.
  258. expect(() => ctx.subagents.start('spawn', structuredRequest(parent, {
  259. outputSchema: { type: 'object', default: () => {} } as unknown as StructuredOutputSchema,
  260. }))).toThrow(/unsupported output schema.*annotation must be JSON data/)
  261. })
  262. it('a post-execute BLOCK on the capture call denies the capture: log and result agree on failure', async () => {
  263. const { ctx, parent, adapter } = await setup([
  264. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 7 }),
  265. textResponse('continues after the blocked capture'),
  266. ])
  267. // A PostToolUse-style hook, registered AFTER the runtime (so the runtime's
  268. // prepend commit listener stays outermost and composes this verdict).
  269. ctx.on('tools/post-execute', (exec, _result, next) => {
  270. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  271. return Promise.resolve({ kind: 'block' as const, feedback: [{ type: 'text' as const, text: 'capture rejected by hook' }] })
  272. }
  273. return next()
  274. })
  275. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  276. const result = await run.result
  277. // No capture was committed: the run reports the schema shortfall...
  278. expect(result.structured).toBeUndefined()
  279. expect(result.stopReason).toBe('error')
  280. // ...the logged tool result is the blocked isError with the feedback...
  281. const child = ctx.agents.get(run.id)!
  282. const results = child.session.events.filter(e => e.type === 'tool/result')
  283. expect((results[0]!.data as { isError?: boolean }).isError).toBe(true)
  284. expect(JSON.stringify((results[0]!.data as { content: unknown }).content)).toContain('capture rejected by hook')
  285. // ...and the turn CONTINUED past the blocked call (no captured veto):
  286. // the model got to react to the failure with a second step.
  287. expect(adapter.requests.length).toBe(2)
  288. await run.dispose()
  289. })
  290. it('a post-execute accept-with-replacement still commits the capture', async () => {
  291. const { ctx, parent } = await setup([
  292. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 8 }),
  293. ])
  294. ctx.on('tools/post-execute', (exec, _result, next) => {
  295. if (exec.name === STRUCTURED_OUTPUT_TOOL) {
  296. return Promise.resolve({ kind: 'accept' as const, content: [{ type: 'text' as const, text: 'recorded (rewritten)' }] })
  297. }
  298. return next()
  299. })
  300. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  301. const result = await run.result
  302. expect(result.stopReason).toBe('completed')
  303. expect(result.structured).toEqual({ answer: 8 })
  304. await run.dispose()
  305. })
  306. it('appends the structured instruction to the child REQUEST\'s system text (base prompt preserved)', async () => {
  307. const { ctx, parent, adapter } = await setup([toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 })])
  308. // A context-wide section stands in for the deployment persona: the
  309. // instruction must APPEND to whatever the prompt pipeline assembled, not
  310. // replace it (AgentOptions has no prompt field — the instruction is
  311. // per-request wire state added by the final-request listener).
  312. ctx.systemPrompt.section({ name: 'test:persona', order: 10, text: 'You are a counter.' })
  313. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  314. await run.result
  315. const childRequest = adapter.requests.at(-1)!
  316. expect(childRequest.system).toContain('You are a counter.')
  317. expect(childRequest.system!.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  318. expect(childRequest.system!.indexOf(STRUCTURED_OUTPUT_INSTRUCTION)).toBeGreaterThan(0)
  319. await run.dispose()
  320. })
  321. it('the instruction rides ONLY structured requests: appended for the child, absent for a plain agent', async () => {
  322. const { ctx, parent, adapter } = await setup([
  323. textResponse('parent answer'),
  324. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  325. ])
  326. parent.send([{ type: 'text', text: 'hello' }])
  327. await parent.whenIdle()
  328. expect(adapter.requests[0]!.system ?? '').not.toContain(STRUCTURED_OUTPUT_INSTRUCTION)
  329. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  330. await run.result
  331. // The loop always assembles a base prompt (the harness identity section),
  332. // so the instruction APPENDS — never replaces.
  333. const childSystem = adapter.requests.at(-1)!.system!
  334. expect(childSystem.endsWith(STRUCTURED_OUTPUT_INSTRUCTION)).toBe(true)
  335. expect(childSystem.length).toBeGreaterThan(STRUCTURED_OUTPUT_INSTRUCTION.length)
  336. await run.dispose()
  337. })
  338. describe('final-request enforcement (the prepend agent/request listener)', () => {
  339. it('a plain agent assembling while the runtime is LIVE gets the placeholder stripped', async () => {
  340. // Run-scoped acquisition means a plain deployment never registers the
  341. // tool at all; the strip branch exists for the CONCURRENT case — a plain
  342. // agent taking a turn while some structured child holds the runtime open.
  343. const { ctx, parent, adapter } = await setup([textResponse('parent answer')])
  344. const hold = acquireStructuredRuntime(ctx)
  345. parent.send([{ type: 'text', text: 'hello' }])
  346. await parent.whenIdle()
  347. // The placeholder IS in the registry during this turn; the assembly the
  348. // loop rendered must not carry it for an agent without a structured run.
  349. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined()
  350. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  351. hold.release()
  352. })
  353. it('a structured child sees structured_output with ITS schema; a plain agent never sees the tool', async () => {
  354. const { ctx, parent, adapter } = await setup([
  355. // Parent turn (a plain agent): must NOT see the tool.
  356. textResponse('parent answer'),
  357. // Child turn: must see it, with the run's schema.
  358. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 42 }),
  359. ])
  360. parent.send([{ type: 'text', text: 'hello' }])
  361. await parent.whenIdle()
  362. expect(toolNames(adapter.requests[0]!)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  363. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  364. await run.result
  365. const childRequest = adapter.requests[1]!
  366. expect(toolNames(childRequest)).toContain(STRUCTURED_OUTPUT_TOOL)
  367. const entry = childRequest.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  368. expect(entry.parameters).toEqual(SCHEMA)
  369. await run.dispose()
  370. })
  371. it('two concurrent structured children each see their OWN schema', async () => {
  372. const otherSchema: StructuredOutputSchema = {
  373. type: 'object',
  374. properties: { verdict: { type: 'string', enum: ['real', 'bogus'] } },
  375. required: ['verdict'],
  376. }
  377. const { ctx, parent, adapter } = await setup([
  378. (options: GenerateOptions) => {
  379. // Answer with whatever schema this child was given — proves each
  380. // request carried the right one regardless of scheduling order.
  381. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  382. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  383. ? { verdict: 'real' }
  384. : { answer: 1 }
  385. return toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, args)
  386. },
  387. (options: GenerateOptions) => {
  388. const entry = options.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!
  389. const args = 'verdict' in (entry.parameters.properties as Record<string, unknown>)
  390. ? { verdict: 'real' }
  391. : { answer: 1 }
  392. return toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, args)
  393. },
  394. ])
  395. const runA = ctx.subagents.start('spawn', structuredRequest(parent))
  396. const runB = ctx.subagents.start('spawn', structuredRequest(parent, { outputSchema: otherSchema }))
  397. const [a, b] = await Promise.all([runA.result, runB.result])
  398. expect(a.structured).toEqual({ answer: 1 })
  399. expect(b.structured).toEqual({ verdict: 'real' })
  400. const schemas = adapter.requests.map(request =>
  401. request.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters)
  402. expect(schemas).toContainEqual(SCHEMA)
  403. expect(schemas).toContainEqual(otherSchema)
  404. await runA.dispose()
  405. await runB.dispose()
  406. })
  407. it('wins against a downstream listener that REPLACES the assembly object', async () => {
  408. const { ctx, parent, adapter } = await setup([
  409. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 5 }),
  410. ])
  411. // A downstream (non-prepend) listener that returns a brand-new assembly —
  412. // the composition caveat that erases cooperative mutations. Registered
  413. // AFTER the runtime's prepend listener, so it runs INSIDE it.
  414. ctx.on('system-prompt/assemble', async (_assembly, _context, next) => {
  415. const replaced = await next()
  416. return { sections: [...replaced.sections], tools: [...replaced.tools], variables: { ...replaced.variables } }
  417. })
  418. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  419. const result = await run.result
  420. expect(result.structured).toEqual({ answer: 5 })
  421. const entry = adapter.requests[0]!.tools!.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)
  422. expect(entry).toBeDefined()
  423. expect(entry!.parameters).toEqual(SCHEMA)
  424. await run.dispose()
  425. })
  426. it('a non-structured agent request keeps tools ABSENT when it had none (no tools: [] materialized)', async () => {
  427. const { parent, adapter } = await setup([
  428. // The registry contributes the placeholder via prompt assembly, so
  429. // tools is an array in the raw request — but after stripping the
  430. // placeholder (its ONLY entry), the field must not be re-added as a
  431. // different shape.
  432. textResponse('plain'),
  433. ])
  434. parent.send([{ type: 'text', text: 'q' }])
  435. await parent.whenIdle()
  436. const request = adapter.requests[0]!
  437. expect(toolNames(request)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  438. await new Promise(resolve => setTimeout(resolve, 0))
  439. })
  440. it('shapes a bare assembly on the waterfall: no-agent context strips the placeholder; a structured agent gains schema + trailing instruction section', async () => {
  441. // Drive ctx.systemPrompt.assemble directly — the enforcement listener
  442. // must tolerate a context with NO agent (a bare diagnostic assemble)
  443. // and shape a structured agent's assembly on the same path the loop
  444. // renders and logs as the request header.
  445. const { ctx, parent } = await setup([])
  446. const acquisition = acquireStructuredRuntime(ctx)
  447. // Bare assemble WHILE the runtime is live: the no-agent branch must
  448. // strip the registered placeholder (before the acquisition there is
  449. // nothing to strip — run-scoped registration).
  450. const bare = await ctx.systemPrompt.assemble({})
  451. expect(bare.tools.map(tool => tool.name)).not.toContain(STRUCTURED_OUTPUT_TOOL)
  452. acquisition.attach(parent, SCHEMA)
  453. const shaped = await ctx.systemPrompt.assemble({ agent: parent })
  454. expect(shaped.tools.map(tool => tool.name)).toContain(STRUCTURED_OUTPUT_TOOL)
  455. expect(shaped.tools.find(tool => tool.name === STRUCTURED_OUTPUT_TOOL)!.parameters).toEqual(SCHEMA)
  456. // The demand travels with the tool: the instruction renders LAST
  457. // (appended post-next(); renderPrompt joins in array order).
  458. expect(shaped.sections.at(-1)).toMatchObject({ name: `tool:${STRUCTURED_OUTPUT_TOOL}`, text: STRUCTURED_OUTPUT_INSTRUCTION })
  459. acquisition.detach(parent)
  460. acquisition.release()
  461. })
  462. })
  463. describe('runtime lifetime (refcount: live structured runs)', () => {
  464. it('the runtime exists exactly while structured runs are live: nothing before, nothing after', async () => {
  465. const { ctx, parent } = await setup([
  466. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 4 }),
  467. ])
  468. // No always-on global state: a context that has run no structured child
  469. // carries no capture tool.
  470. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  471. const run = ctx.subagents.start('spawn', structuredRequest(parent))
  472. const result = await run.result
  473. // The capture succeeded — the registrations existed while the run lived.
  474. expect(result.structured).toEqual({ answer: 4 })
  475. // The run's settle released the last acquisition.
  476. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  477. await run.dispose()
  478. })
  479. it('concurrent structured runs share one runtime; the last settle disposes it', async () => {
  480. const { ctx, parent } = await setup([
  481. toolCallResponse('c1', STRUCTURED_OUTPUT_TOOL, { answer: 1 }),
  482. toolCallResponse('c2', STRUCTURED_OUTPUT_TOOL, { answer: 2 }),
  483. ])
  484. const first = ctx.subagents.start('spawn', structuredRequest(parent))
  485. const second = ctx.subagents.start('spawn', structuredRequest(parent))
  486. const [a, b] = await Promise.all([first.result, second.result])
  487. expect([a.structured, b.structured].sort()).toEqual([{ answer: 1 }, { answer: 2 }].sort())
  488. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  489. await first.dispose()
  490. await second.dispose()
  491. })
  492. it('acquisition release is idempotent (double release cannot underflow the refcount)', async () => {
  493. const ctx = new Context()
  494. await ctx.plugin(SystemPrompt)
  495. await ctx.plugin(ToolRegistry)
  496. const first = acquireStructuredRuntime(ctx)
  497. const second = acquireStructuredRuntime(ctx)
  498. first.release()
  499. first.release()
  500. // The second holder still keeps the tool registered.
  501. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined()
  502. second.release()
  503. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  504. })
  505. it('registers the capture tool through the scoped fiber when tools loads after the acquisition', async () => {
  506. // The Loader starts sibling plugins concurrently, so a backend can
  507. // acquire the runtime before dsh-tools has applied. The capture tool
  508. // must then register as soon as `tools` exists — via the inject fiber,
  509. // not by deferring the backend (which would reorder the prompt's tools).
  510. const ctx = new Context()
  511. const acquisition = acquireStructuredRuntime(ctx)
  512. await ctx.plugin(SystemPrompt)
  513. await ctx.plugin(ToolRegistry)
  514. // Fiber activation completes asynchronously after the service appears.
  515. await new Promise(resolve => setImmediate(resolve))
  516. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeDefined()
  517. acquisition.release()
  518. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  519. })
  520. it('releasing before tools ever loads disposes the pending fiber without registering', async () => {
  521. const ctx = new Context()
  522. const acquisition = acquireStructuredRuntime(ctx)
  523. acquisition.release()
  524. await ctx.plugin(SystemPrompt)
  525. await ctx.plugin(ToolRegistry)
  526. await new Promise(resolve => setImmediate(resolve))
  527. // The disposed fiber never fires: nothing registers after the fact.
  528. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  529. })
  530. it('attach/captured/detach manage per-agent state through the acquisition surface', async () => {
  531. const { ctx, parent } = await setup([])
  532. const acquisition = acquireStructuredRuntime(ctx)
  533. expect(acquisition.captured(parent)).toBeUndefined()
  534. acquisition.attach(parent, SCHEMA)
  535. expect(acquisition.captured(parent)).toBeUndefined()
  536. acquisition.detach(parent)
  537. acquisition.detach(parent)
  538. acquisition.release()
  539. // That manual acquisition was the ONLY holder - release disposes.
  540. expect(ctx.tools.get(STRUCTURED_OUTPUT_TOOL)).toBeUndefined()
  541. })
  542. })
  543. it('a direct structured_output call from an agent WITHOUT a structured run is an isError', async () => {
  544. const { ctx, parent } = await setup([])
  545. // Hold the runtime open (run-scoped: nothing is registered otherwise) so
  546. // the call reaches the capture tool's own fail-loud guard, not UNKNOWN_TOOL.
  547. const hold = acquireStructuredRuntime(ctx)
  548. const result = await ctx.tools.execute({
  549. callId: 'x' as never,
  550. name: STRUCTURED_OUTPUT_TOOL,
  551. arguments: { answer: 1 },
  552. agent: parent,
  553. })
  554. expect(result.isError).toBe(true)
  555. expect(JSON.stringify(result.content)).toContain('only available to subagents')
  556. hold.release()
  557. })
  558. it('a structured_output call with NO calling agent at all is an isError', async () => {
  559. const { ctx } = await setup([])
  560. const hold = acquireStructuredRuntime(ctx)
  561. const result = await ctx.tools.execute({
  562. callId: 'x' as never,
  563. name: STRUCTURED_OUTPUT_TOOL,
  564. arguments: { answer: 1 },
  565. })
  566. expect(result.isError).toBe(true)
  567. hold.release()
  568. })
  569. })