loop.spec.ts 43 KB

1234567891011121314151617181920212223242526272829303132333435363738394041424344454647484950515253545556575859606162636465666768697071727374757677787980818283848586878889909192939495969798991001011021031041051061071081091101111121131141151161171181191201211221231241251261271281291301311321331341351361371381391401411421431441451461471481491501511521531541551561571581591601611621631641651661671681691701711721731741751761771781791801811821831841851861871881891901911921931941951961971981992002012022032042052062072082092102112122132142152162172182192202212222232242252262272282292302312322332342352362372382392402412422432442452462472482492502512522532542552562572582592602612622632642652662672682692702712722732742752762772782792802812822832842852862872882892902912922932942952962972982993003013023033043053063073083093103113123133143153163173183193203213223233243253263273283293303313323333343353363373383393403413423433443453463473483493503513523533543553563573583593603613623633643653663673683693703713723733743753763773783793803813823833843853863873883893903913923933943953963973983994004014024034044054064074084094104114124134144154164174184194204214224234244254264274284294304314324334344354364374384394404414424434444454464474484494504514524534544554564574584594604614624634644654664674684694704714724734744754764774784794804814824834844854864874884894904914924934944954964974984995005015025035045055065075085095105115125135145155165175185195205215225235245255265275285295305315325335345355365375385395405415425435445455465475485495505515525535545555565575585595605615625635645655665675685695705715725735745755765775785795805815825835845855865875885895905915925935945955965975985996006016026036046056066076086096106116126136146156166176186196206216226236246256266276286296306316326336346356366376386396406416426436446456466476486496506516526536546556566576586596606616626636646656666676686696706716726736746756766776786796806816826836846856866876886896906916926936946956966976986997007017027037047057067077087097107117127137147157167177187197207217227237247257267277287297307317327337347357367377387397407417427437447457467477487497507517527537547557567577587597607617627637647657667677687697707717727737747757767777787797807817827837847857867877887897907917927937947957967977987998008018028038048058068078088098108118128138148158168178188198208218228238248258268278288298308318328338348358368378388398408418428438448458468478488498508518528538548558568578588598608618628638648658668678688698708718728738748758768778788798808818828838848858868878888898908918928938948958968978988999009019029039049059069079089099109119129139149159169179189199209219229239249259269279289299309319329339349359369379389399409419429439449459469479489499509519529539549559569579589599609619629639649659669679689699709719729739749759769779789799809819829839849859869879889899909919929939949959969979989991000
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { CallId, StreamChunk } from '@deepseek-ai/dsh-llm'
  4. import SessionStore, { SessionId, TurnEndReason } from '@deepseek-ai/dsh-session'
  5. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  6. import ToolRegistry, { defineTool } from '@deepseek-ai/dsh-tools'
  7. import AgentRegistry, { type Agent } from '@deepseek-ai/dsh-agent'
  8. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  9. import { MockAdapter, maxTokensResponse, textResponse, toolCallResponse } from './mock-adapter.ts'
  10. function driverDone(agent: Agent): Promise<void> {
  11. return (agent as Agent & { done: Promise<void> }).done
  12. }
  13. async function harness(adapter: MockAdapter, persona = '') {
  14. const ctx = new Context()
  15. await ctx.plugin(LlmService)
  16. await ctx.plugin(SessionStore)
  17. await ctx.plugin(SystemPrompt, { persona })
  18. await ctx.plugin(ToolRegistry)
  19. await ctx.plugin(AgentRegistry)
  20. await ctx.plugin(AgentLoop, { agents: [] })
  21. ctx.llm.registerAdapter(['mock'], adapter)
  22. return ctx
  23. }
  24. /**
  25. * Wait for the agent's NEXT transition to idle. Always event-based: callers
  26. * invoke this right after send(), when the loop hasn't woken yet (status is
  27. * still 'idle' synchronously), so polling the current status would lie.
  28. */
  29. function waitForIdle(ctx: Context, agent: Agent): Promise<void> {
  30. return new Promise((resolve) => {
  31. const dispose = ctx.on('agent/status', (subject, status) => {
  32. if (subject === agent && status === 'idle') {
  33. dispose()
  34. resolve()
  35. }
  36. })
  37. })
  38. }
  39. function send(agent: Agent, text: string) {
  40. agent.send([{ type: 'text', text }])
  41. }
  42. describe('agent loop', () => {
  43. it('runs a simple turn: queued message → model → idle, with ordered events', async () => {
  44. const adapter = new MockAdapter([textResponse('hello there')])
  45. const ctx = await harness(adapter)
  46. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  47. // All boundaries — turn and step — are durable session events on the
  48. // session/event feed (no agent/* mirror). Record them in fire order to
  49. // assert the full boundary nesting.
  50. const order: string[] = []
  51. ctx.on('session/event', (_session, event) => {
  52. if (event.type === 'turn/start' || event.type === 'step/start' || event.type === 'step/end' || event.type === 'turn/end') {
  53. order.push(event.type)
  54. }
  55. })
  56. send(agent, 'hi')
  57. await waitForIdle(ctx, agent)
  58. expect(order).toEqual(['turn/start', 'step/start', 'step/end', 'turn/end'])
  59. const types = agent.session.events.map(e => e.type)
  60. // turn/start opens the turn, THEN the queued user message is recorded inside
  61. // it (every event is turn-enclosed), then the assembled message (carrying the
  62. // step's usage).
  63. expect(types[0]).toBe('turn/start')
  64. expect(types[1]).toBe('user/message')
  65. expect(types).toContain('assistant/message')
  66. const assistantMessage = agent.session.events.find(e => e.type === 'assistant/message')
  67. expect(assistantMessage?.type === 'assistant/message' && assistantMessage.data.usage).toEqual({ inputTokens: 10, outputTokens: 'hello there'.length })
  68. expect(types.at(-1)).toBe('turn/end')
  69. // derived history: user + assistant
  70. const messages = agent.session.deriveMessages()
  71. expect(messages.map(m => m.role)).toEqual(['user', 'assistant'])
  72. expect(messages[1]!.content).toEqual([{ type: 'text', text: 'hello there' }])
  73. })
  74. it('round-trips tool calls: model requests tool → executes → result in next request', async () => {
  75. const adapter = new MockAdapter([
  76. toolCallResponse('c1', 'echo', { text: 'ping' }, 'calling echo'),
  77. textResponse('done'),
  78. ])
  79. const ctx = await harness(adapter)
  80. ctx.tools.register(defineTool({
  81. name: 'echo',
  82. description: 'echo back',
  83. parameters: { text: { type: 'string' } },
  84. async execute(args) {
  85. return [{ type: 'text', text: `echo: ${args.text}` }]
  86. },
  87. }))
  88. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  89. send(agent, 'use the tool')
  90. await waitForIdle(ctx, agent)
  91. // two model calls happened (tool-call step, then final step)
  92. expect(adapter.requests).toHaveLength(2)
  93. // the second request's derived history contains the tool result
  94. const secondMessages = adapter.requests[1]!.messages
  95. const toolResultMessage = secondMessages.find(m =>
  96. m.content.some(b => b.type === 'tool-result'))
  97. expect(toolResultMessage).toBeDefined()
  98. const block = toolResultMessage!.content.find(b => b.type === 'tool-result')!
  99. expect(block).toMatchObject({ toolCallId: 'c1', isError: false })
  100. expect((block).content).toEqual([{ type: 'text', text: 'echo: ping' }])
  101. // session log records call + result
  102. const types = agent.session.events.map(e => e.type)
  103. expect(types).toContain('tool/call')
  104. expect(types).toContain('tool/result')
  105. })
  106. it('threads a tool-attached meta (execute object return) onto the tool/result event', async () => {
  107. const adapter = new MockAdapter([
  108. toolCallResponse('c1', 'writer', { path: 'a.txt' }, 'writing'),
  109. textResponse('done'),
  110. ])
  111. const ctx = await harness(adapter)
  112. // A tool that returns the { content, meta } object form: the loop must
  113. // persist `meta` on the tool/result event so a UI reproduces the card on replay.
  114. ctx.tools.register(defineTool({
  115. name: 'writer',
  116. description: 'writes a file',
  117. parameters: { path: { type: 'string' } },
  118. async execute() {
  119. return { content: [{ type: 'text', text: 'ok' }], meta: { diffs: [{ path: 'a.txt', oldText: null, newText: 'x' }] } }
  120. },
  121. }))
  122. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  123. send(agent, 'use the tool')
  124. await waitForIdle(ctx, agent)
  125. const toolResult = agent.session.events.find(e => e.type === 'tool/result')
  126. expect(toolResult?.type === 'tool/result' && toolResult.data.meta)
  127. .toEqual({ diffs: [{ path: 'a.txt', oldText: null, newText: 'x' }] })
  128. })
  129. it('renders harness identity, then the persona, then tool guidance — with {{variables}} resolved', async () => {
  130. const adapter = new MockAdapter([textResponse('ok')])
  131. // The persona is a TEMPLATE: {{model}} is the loop-registered variable
  132. // projecting this agent's configured model, so the model knows its own name.
  133. const ctx = await harness(adapter, 'You are a test agent on {{model}}.')
  134. ctx.systemPrompt.section({ name: 'tool:noop', order: 100, text: 'Use the noop tool wisely.' })
  135. ctx.tools.register(defineTool({
  136. name: 'noop',
  137. description: 'does nothing',
  138. parameters: {},
  139. async execute() {
  140. return []
  141. },
  142. }))
  143. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  144. send(agent, 'hi')
  145. await waitForIdle(ctx, agent)
  146. const request = adapter.requests[0]
  147. expect(request!.system).toBe('You are an AI agent powered by the DeepSeek Harness SDK.\n\nYou are a test agent on mock.\n\nUse the noop tool wisely.')
  148. expect(request!.tools?.map(t => t.name)).toEqual(['noop'])
  149. })
  150. it('resolves {{cwd}} from the agent session workspace (factory create with meta.cwd)', async () => {
  151. const adapter = new MockAdapter([textResponse('ok')])
  152. const ctx = await harness(adapter, 'Working in {{cwd}}.')
  153. const handle = await ctx.agents.create({
  154. sessionId: SessionId('s-cwd'),
  155. meta: { cwd: '/work/space' },
  156. agentOptions: { model: 'mock' },
  157. })
  158. const agent = handle.agent
  159. send(agent, 'hi')
  160. await waitForIdle(ctx, agent)
  161. expect(adapter.requests[0]!.system).toBe('You are an AI agent powered by the DeepSeek Harness SDK.\n\nWorking in /work/space.')
  162. })
  163. it('contains a strict-variable render failure: the turn errors, the loop keeps serving turns', async () => {
  164. // A persona claiming {{cwd}} on a session with NO cwd is a deployment
  165. // authoring error — renderPrompt throws, the turn ends with an error, and
  166. // the same agent must then RUN a later turn to completion (not merely
  167. // report idle status): a rescue listener supplies the variable and the
  168. // follow-up prompt reaches the model.
  169. const adapter = new MockAdapter([textResponse('ok after rescue')])
  170. const ctx = await harness(adapter, 'In {{cwd}}.')
  171. const errors: Error[] = []
  172. ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error))
  173. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  174. send(agent, 'hi')
  175. await waitForIdle(ctx, agent)
  176. expect(adapter.requests).toHaveLength(0) // the request was never sent
  177. expect(errors.some(e => e.message.includes('no value for this assembly'))).toBe(true)
  178. const turnEnd = agent.session.events.find(e => e.type === 'turn/end')
  179. expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason.kind).toBe('error')
  180. // The loop survived: a waterfall listener rescues {{cwd}} and the SAME
  181. // agent completes a real model turn.
  182. ctx.on('system-prompt/assemble', async (assembly, _context, next) => {
  183. assembly.variables['cwd'] = '/rescued'
  184. return next()
  185. })
  186. send(agent, 'again')
  187. await waitForIdle(ctx, agent)
  188. expect(adapter.requests).toHaveLength(1)
  189. expect(adapter.requests[0]!.system).toBe('You are an AI agent powered by the DeepSeek Harness SDK.\n\nIn /rescued.')
  190. const turnEnds = agent.session.events.filter(e => e.type === 'turn/end')
  191. expect(turnEnds).toHaveLength(2)
  192. expect(turnEnds[1]?.type === 'turn/end' && turnEnds[1].data.reason.kind).toBe('completed')
  193. })
  194. it('supports the model-via-agent/request path with a {{model}} persona: the supplier states it via the assemble waterfall', async () => {
  195. // AgentOptions.model unset: the model arrives in the agent/request
  196. // waterfall (the loop's documented fallback — see runStep's no-model
  197. // error). {{model}} renders BEFORE that waterfall, so the SAME plugin
  198. // states the fact early on system-prompt/assemble — the owner of a
  199. // late-bound fact owns stating it wherever it is claimed.
  200. const adapter = new MockAdapter([textResponse('ok')])
  201. const ctx = await harness(adapter, 'You run on {{model}}.')
  202. ctx.on('system-prompt/assemble', async (assembly, _context, next) => {
  203. assembly.variables['model'] = 'mock'
  204. return next()
  205. })
  206. ctx.on('agent/request', async (_agent, _turn, _step, config, _next) => {
  207. return { ...config, model: 'mock' }
  208. })
  209. const agent = ctx.agentLoop.create(SessionId('a-late-model'), {})
  210. send(agent, 'hi')
  211. await waitForIdle(ctx, agent)
  212. expect(adapter.requests).toHaveLength(1)
  213. expect(adapter.requests[0]!.model).toBe('mock')
  214. expect(adapter.requests[0]!.system).toBe('You are an AI agent powered by the DeepSeek Harness SDK.\n\nYou run on mock.')
  215. })
  216. it.each([
  217. ['BigInt', { n: 1n }],
  218. ['Map', new Map([['key', 'value']])],
  219. ['class instance', new (class ResultMeta { x = 1 })()],
  220. ])('normalizes non-JSON tool meta (%s) before the durable result commit', async (_kind, meta) => {
  221. const adapter = new MockAdapter([
  222. toolCallResponse('bad-meta-call', 'bad-meta', {}, 'calling'),
  223. textResponse('recovered'),
  224. ])
  225. const ctx = await harness(adapter)
  226. ctx.tools.register(defineTool({
  227. name: 'bad-meta',
  228. description: 'returns invalid durable metadata',
  229. parameters: {},
  230. execute: () => Promise.resolve({ content: [{ type: 'text' as const, text: 'apparent success' }], meta }),
  231. }))
  232. const agent = ctx.agentLoop.create(SessionId('bad-meta-agent'), { model: 'mock' })
  233. send(agent, 'use the tool')
  234. await waitForIdle(ctx, agent)
  235. const result = agent.session.events.find(event => event.type === 'tool/result')
  236. expect(result?.type).toBe('tool/result')
  237. if (result?.type === 'tool/result') {
  238. expect(result.data.callId).toBe('bad-meta-call')
  239. expect(result.data.isError).toBe(true)
  240. expect(result.data.meta).toBeUndefined()
  241. expect(result.data.content).toEqual([{
  242. type: 'text',
  243. text: 'Error: tool result must be losslessly JSON-serializable',
  244. }])
  245. }
  246. // The normalized failure was durably logged and fed back to the model; the
  247. // turn continued normally instead of failing after an apparent success.
  248. expect(adapter.requests).toHaveLength(2)
  249. expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('losslessly JSON-serializable')
  250. })
  251. it('omits the system field when a system-prompt/assemble veto empties the assembly', async () => {
  252. // The documented escape valve: a deployment that must drop the harness
  253. // openers short-circuits the assemble waterfall; the request then carries
  254. // NO system field at all (not an empty string).
  255. const adapter = new MockAdapter([textResponse('ok')])
  256. const ctx = await harness(adapter)
  257. ctx.on('system-prompt/assemble', async () => ({ sections: [], tools: [], variables: {} }))
  258. const agent = ctx.agentLoop.create(SessionId('a-no-system'), { model: 'mock' })
  259. send(agent, 'hi')
  260. await waitForIdle(ctx, agent)
  261. expect(adapter.requests).toHaveLength(1)
  262. expect('system' in adapter.requests[0]!).toBe(false)
  263. })
  264. it('records raw chunks for replay as assistant/chunk session events', async () => {
  265. const adapter = new MockAdapter([textResponse('abc')])
  266. const ctx = await harness(adapter)
  267. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  268. send(agent, 'hi')
  269. await waitForIdle(ctx, agent)
  270. const chunkEvents = agent.session.events.filter(e => e.type === 'assistant/chunk')
  271. // textResponse('abc') = block-start + 3 deltas + block-end + usage + finish = 7
  272. expect(chunkEvents).toHaveLength(7)
  273. // replay: chunk events alone re-assemble to the recorded assistant message
  274. const deltaText = chunkEvents
  275. .flatMap(e => e.type === 'assistant/chunk' ? [e.data.chunk] : [])
  276. .filter((c: StreamChunk): c is Extract<StreamChunk, { type: 'text-delta' }> => c.type === 'text-delta')
  277. .map(c => c.text)
  278. .join('')
  279. expect(deltaText).toBe('abc')
  280. })
  281. it('injects steering between steps and continues the turn', async () => {
  282. const adapter = new MockAdapter([
  283. toolCallResponse('c1', 'slow', {}),
  284. textResponse('addressed the steering'),
  285. ])
  286. const ctx = await harness(adapter)
  287. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  288. ctx.tools.register(defineTool({
  289. name: 'slow',
  290. description: '',
  291. parameters: {},
  292. async execute() {
  293. // steer while the turn is running (during tool execution)
  294. agent.steer([{ type: 'text', text: 'change of plans' }])
  295. return [{ type: 'text', text: 'tool done' }]
  296. },
  297. }))
  298. send(agent, 'start')
  299. await waitForIdle(ctx, agent)
  300. const types = agent.session.events.map(e => e.type)
  301. expect(types).toContain('steering/message')
  302. // steering recorded before the second step's request derived its history
  303. const steeringSeq = agent.session.events.find(e => e.type === 'steering/message')!.seq
  304. const secondStepStart = agent.session.events.filter(e => e.type === 'step/start')[1]
  305. expect(secondStepStart).toBeDefined()
  306. expect(steeringSeq).toBeLessThan(secondStepStart!.seq)
  307. // the second model request saw the steering content
  308. const secondRequest = adapter.requests[1]
  309. const flat = JSON.stringify(secondRequest!.messages)
  310. expect(flat).toContain('change of plans')
  311. })
  312. it('steering while idle behaves like send (starts a turn)', async () => {
  313. const adapter = new MockAdapter([textResponse('ok')])
  314. const ctx = await harness(adapter)
  315. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  316. agent.steer([{ type: 'text', text: 'hello' }])
  317. await waitForIdle(ctx, agent)
  318. expect(agent.session.events.some(e => e.type === 'user/message')).toBe(true)
  319. })
  320. it('inject() while idle wraps context in a one-shot turn, visible to the next request', async () => {
  321. const adapter = new MockAdapter([textResponse('ok')])
  322. const ctx = await harness(adapter)
  323. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  324. agent.inject([{ type: 'text', text: 'file changed: a.ts' }], { source: { kind: 'plugin', plugin: 'watcher' } })
  325. // The idle inject records a self-contained turn (turn/start → context/message
  326. // → turn/end) so the event stays turn-enclosed, but does NOT run the model.
  327. await new Promise(r => setTimeout(r, 20))
  328. expect(agent.status).toBe('idle')
  329. expect(adapter.requests).toHaveLength(0)
  330. const injectedTurn = agent.session.events.filter(e => e.type === 'turn/start')
  331. expect(injectedTurn).toHaveLength(1)
  332. const it0 = injectedTurn[0]!
  333. expect(it0.type === 'turn/start' && it0.data.trigger.kind).toBe('injection')
  334. expect(agent.session.events.at(-1)!.type).toBe('turn/end') // turn-enclosed
  335. send(agent, 'go')
  336. await waitForIdle(ctx, agent)
  337. const flat = JSON.stringify(adapter.requests[0]!.messages)
  338. expect(flat).toContain('file changed: a.ts')
  339. expect(flat).toContain('<context source=\\"plugin\\">')
  340. })
  341. it('inject() while running appends into the open turn (no extra synthetic turn)', async () => {
  342. const adapter = new MockAdapter([
  343. toolCallResponse('c1', 'noticer', {}, 'calling'),
  344. textResponse('done'),
  345. ])
  346. const ctx = await harness(adapter)
  347. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  348. // A tool that injects mid-execution: at this point the agent is running, so
  349. // inject must append the context/message into the ALREADY-open turn rather
  350. // than wrap it in its own one-shot turn.
  351. ctx.tools.register(defineTool({
  352. name: 'noticer',
  353. description: 'injects a notice',
  354. parameters: {},
  355. async execute() {
  356. agent.inject([{ type: 'text', text: 'mid-turn notice' }], { source: { kind: 'plugin', plugin: 'x' } })
  357. return [{ type: 'text', text: 'ok' }]
  358. },
  359. }))
  360. send(agent, 'go')
  361. await waitForIdle(ctx, agent)
  362. // Exactly ONE turn ran (no synthetic injection turn), and the mid-turn
  363. // context/message sits inside it.
  364. const turnStarts = agent.session.events.filter(e => e.type === 'turn/start')
  365. expect(turnStarts).toHaveLength(1)
  366. const ts0 = turnStarts[0]!
  367. expect(ts0.type === 'turn/start' && ts0.data.trigger.kind).toBe('message')
  368. expect(agent.session.events.some(e => e.type === 'context/message')).toBe(true)
  369. })
  370. it('agent/turn-continuation can force-continue (/loop pattern) and force-stop', async () => {
  371. // force-continue: model never calls tools, but a plugin forces 3 steps
  372. const adapter = new MockAdapter([
  373. textResponse('step 1'),
  374. textResponse('step 2'),
  375. textResponse('step 3'),
  376. ])
  377. const ctx = await harness(adapter)
  378. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  379. let steps = 0
  380. ctx.on('session/event', (_session, event) => { if (event.type === 'step/end') steps++ })
  381. ctx.on('agent/turn-continuation', async (_agent, _turn, _defaultDecision, next) => {
  382. if (steps < 3) return { action: 'continue' as const }
  383. return next()
  384. })
  385. send(agent, 'go')
  386. await waitForIdle(ctx, agent)
  387. expect(steps).toBe(3)
  388. expect(adapter.requests).toHaveLength(3)
  389. })
  390. it('agent/turn-continuation can veto continuation despite tool calls (budget-guard pattern)', async () => {
  391. const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'x' })])
  392. const ctx = await harness(adapter)
  393. ctx.tools.register(defineTool({
  394. name: 'echo',
  395. description: '',
  396. parameters: { text: { type: 'string' } },
  397. async execute(args) {
  398. return [{ type: 'text', text: String(args.text) }]
  399. },
  400. }))
  401. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  402. ctx.on('agent/turn-continuation', async () => ({ action: 'stop' }) as const)
  403. send(agent, 'go')
  404. await waitForIdle(ctx, agent)
  405. // only one model call despite the tool call requesting a follow-up
  406. expect(adapter.requests).toHaveLength(1)
  407. // tool still executed before the decision
  408. expect(agent.session.events.some(e => e.type === 'tool/result')).toBe(true)
  409. })
  410. it('agent/request waterfall switches models by returning a replacement config; the switch is logged', async () => {
  411. const adapter = new MockAdapter([textResponse('ok')])
  412. const ctx = await harness(adapter)
  413. ctx.llm.registerAdapter(['other-model'], adapter)
  414. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  415. ctx.on('agent/request', async (_agent, _turn, _step, config, _next) => {
  416. // The seed is frozen — config is not a mutable per-call knob; a switch
  417. // is proposed by returning a replacement, and the loop logs it.
  418. expect(Object.isFrozen(config)).toBe(true)
  419. expect(() => { (config as { model: string }).model = 'other-model' }).toThrow(TypeError)
  420. return { ...config, model: 'other-model' }
  421. })
  422. send(agent, 'hi')
  423. await waitForIdle(ctx, agent)
  424. expect(adapter.requests[0]!.model).toBe('other-model')
  425. // The header event records what the request ACTUALLY used — the switch is
  426. // a reconstructable fact, not silent drift.
  427. const headerEvent = agent.session.events.find(e => e.type === 'request/header')
  428. expect(headerEvent?.type === 'request/header' && headerEvent.data.header.config.model).toBe('other-model')
  429. })
  430. it('agent/pre-step fires once per step before the step is opened', async () => {
  431. // Two steps (a tool call, then a final text turn) → two model calls → two
  432. // pre-step fires, each carrying the assembled full system prompt, BEFORE
  433. // the step is opened and its request is derived (the request the adapter
  434. // sees reflects any surface state at fire time).
  435. const adapter = new MockAdapter([
  436. toolCallResponse('c1', 'echo', {}, 'calling echo'),
  437. textResponse('done'),
  438. ])
  439. const ctx = await harness(adapter)
  440. ctx.tools.register(defineTool({
  441. name: 'echo', description: 'echo', parameters: {},
  442. async execute() { return [{ type: 'text', text: 'echoed' }] },
  443. }))
  444. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  445. const fires: { turn: number; step: number; fullSystemPrompt: string }[] = []
  446. ctx.on('agent/pre-step', (subject, turn, step, fullSystemPrompt) => {
  447. if (subject === agent) fires.push({ turn, step, fullSystemPrompt })
  448. })
  449. send(agent, 'go')
  450. await waitForIdle(ctx, agent)
  451. // One fire per step, in order, each with the assembled system prompt
  452. // (here just the loop's own harness-identity section — no persona set).
  453. const HARNESS = 'You are an AI agent powered by the DeepSeek Harness SDK.'
  454. expect(fires).toEqual([
  455. { turn: 1, step: 1, fullSystemPrompt: HARNESS },
  456. { turn: 1, step: 2, fullSystemPrompt: HARNESS },
  457. ])
  458. })
  459. it('agent/pre-step fires BEFORE the step it precedes opens (events land outside the step)', async () => {
  460. // A listener appending a surface node in pre-step lands it BEFORE step/start
  461. // in the log — proving the seam fires outside the step. The node is still in
  462. // the derived request for that step (derive happens after step/start).
  463. const adapter = new MockAdapter([textResponse('ok')])
  464. const ctx = await harness(adapter)
  465. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  466. let injected = false
  467. ctx.on('agent/pre-step', (subject) => {
  468. if (subject === agent && !injected) {
  469. injected = true
  470. subject.session.append('context/message', {
  471. content: [{ type: 'text', text: 'INJECTED-IN-PRE-STEP' }],
  472. source: { kind: 'plugin', plugin: 'test' },
  473. }, { surfaceOp: 'append' })
  474. }
  475. })
  476. send(agent, 'go')
  477. await waitForIdle(ctx, agent)
  478. // The adapter's request includes the node injected during pre-step (derive
  479. // reflects it).
  480. const text = JSON.stringify(adapter.requests[0]!.messages)
  481. expect(text).toContain('INJECTED-IN-PRE-STEP')
  482. // And the injected event sits BEFORE the first step/start in the log —
  483. // the seam fired outside the step.
  484. const events = agent.session.events
  485. const injectedSeq = events.find(e => e.type === 'context/message')!.seq
  486. const firstStepStartSeq = events.find(e => e.type === 'step/start')!.seq
  487. expect(injectedSeq).toBeLessThan(firstStepStartSeq)
  488. })
  489. it('a throwing agent/pre-step listener ends the turn (error), not the loop', async () => {
  490. // The seam fires before step/start, so a throw escapes to runTurn's outer
  491. // catch: the not-yet-open step closes as a no-op, the failure surfaces via
  492. // agent/error, and the turn ends `error` (recorded on the durable turn/end).
  493. // The loop survives and a follow-up prompt still runs.
  494. const adapter = new MockAdapter([textResponse('second turn ok')])
  495. const ctx = await harness(adapter)
  496. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  497. let throwOnce = true
  498. ctx.on('agent/pre-step', () => {
  499. if (throwOnce) { throwOnce = false; throw new Error('boom in pre-step') }
  500. })
  501. const errors: Error[] = []
  502. ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error))
  503. send(agent, 'first')
  504. await waitForIdle(ctx, agent)
  505. // The first turn failed at step 1 (no model call happened), surfaced via
  506. // agent/error, with the durable failure on turn/end.reason.
  507. expect(errors).toHaveLength(1)
  508. expect(errors[0]!.message).toContain('boom in pre-step')
  509. expect(adapter.requests.length).toBe(0)
  510. const firstTurnEnd = agent.session.events.find(e => e.type === 'turn/end')
  511. expect(firstTurnEnd?.type === 'turn/end' && firstTurnEnd.data.reason).toMatchObject({ kind: 'error', step: 1 })
  512. // The step opened-and-closed count stays balanced even though it never ran.
  513. const types = agent.session.events.map(e => e.type)
  514. expect(types.filter(t => t === 'step/start').length).toBe(types.filter(t => t === 'step/end').length)
  515. // The loop survived: a second prompt runs a normal completed turn.
  516. send(agent, 'second')
  517. await waitForIdle(ctx, agent)
  518. expect(adapter.requests.length).toBe(1)
  519. const lastTurnEnd = agent.session.events.findLast(e => e.type === 'turn/end')
  520. expect(lastTurnEnd?.type === 'turn/end' && lastTurnEnd.data.reason).toEqual({ kind: 'completed' })
  521. })
  522. it('cancel() mid-stream ends the turn with reason aborted', async () => {
  523. const adapter = new MockAdapter(['hang'])
  524. const ctx = await harness(adapter)
  525. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  526. const reasons: TurnEndReason[] = []
  527. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  528. send(agent, 'go')
  529. // wait until the stream is hanging, then cancel
  530. await new Promise(r => setTimeout(r, 30))
  531. expect(agent.status).toBe('running')
  532. agent.cancel('user interrupt')
  533. await waitForIdle(ctx, agent)
  534. expect(reasons).toEqual([{ kind: 'aborted', reason: 'user interrupt' }])
  535. })
  536. it('surfaces max-tokens as the turn-end reason when the last step is cut off', async () => {
  537. // A single step that ends with a max-tokens finish (no tool calls): the
  538. // turn stops by default and ends max-tokens, not completed.
  539. const adapter = new MockAdapter([maxTokensResponse('truncat')])
  540. const ctx = await harness(adapter)
  541. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  542. const reasons: TurnEndReason[] = []
  543. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  544. send(agent, 'go')
  545. await waitForIdle(ctx, agent)
  546. expect(adapter.requests).toHaveLength(1)
  547. expect(reasons).toEqual([{ kind: 'max-tokens' }])
  548. // and the reason is recorded in the log's turn/end event
  549. const turnEnd = agent.session.events.findLast(e => e.type === 'turn/end')
  550. expect(turnEnd!.data.reason).toEqual({ kind: 'max-tokens' })
  551. })
  552. it('a max-tokens step earlier in a turn still surfaces as max-tokens after a later completed step', async () => {
  553. // Step 1 is cut off (max-tokens, no tool calls → would stop by default), so
  554. // continuation must be FORCED to reach step 2 which finishes normally
  555. // (stop). The rule "any max-tokens step surfaces as max-tokens" means the
  556. // turn ends max-tokens even though the LAST step completed cleanly.
  557. const adapter = new MockAdapter([
  558. maxTokensResponse('first half'),
  559. textResponse('second half'),
  560. ])
  561. const ctx = await harness(adapter)
  562. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  563. let steps = 0
  564. ctx.on('session/event', (_session, event) => { if (event.type === 'step/end') steps++ })
  565. // Force exactly one continuation (step 1 → step 2), then defer to default
  566. // (step 2 is a plain stop with no tool calls → stops).
  567. ctx.on('agent/turn-continuation', async (_agent, _turn, _defaultDecision, next) => {
  568. if (steps < 2) return { action: 'continue' as const }
  569. return next()
  570. })
  571. const reasons: TurnEndReason[] = []
  572. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  573. send(agent, 'go')
  574. await waitForIdle(ctx, agent)
  575. expect(steps).toBe(2)
  576. expect(adapter.requests).toHaveLength(2)
  577. expect(adapter.requests[1]!.messages).toEqual([
  578. { role: 'user', content: [{ type: 'text', text: 'go' }] },
  579. { role: 'assistant', content: [{ type: 'text', text: 'first half' }] },
  580. ])
  581. expect(reasons).toEqual([{ kind: 'max-tokens' }])
  582. })
  583. it('a completed step after no max-tokens keeps the turn completed (max-tokens does not leak across turns)', async () => {
  584. // Two consecutive turns: turn 1 is cut off (max-tokens), turn 2 is a clean
  585. // stop. The per-turn reason must be independent — turn 2 ends completed.
  586. const adapter = new MockAdapter([maxTokensResponse('cut'), textResponse('clean')])
  587. const ctx = await harness(adapter)
  588. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  589. const reasons: TurnEndReason[] = []
  590. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  591. send(agent, 'first')
  592. await waitForIdle(ctx, agent)
  593. send(agent, 'second')
  594. await waitForIdle(ctx, agent)
  595. expect(reasons).toEqual([{ kind: 'max-tokens' }, { kind: 'completed' }])
  596. })
  597. it('does not dispatch tool calls from a max-tokens-truncated step', async () => {
  598. const callId = CallId('c1')
  599. const adapter = new MockAdapter([[
  600. { type: 'block-start', index: 0, blockType: 'tool-call' },
  601. { type: 'tool-call-delta', index: 0, id: callId, name: 'echo', argumentsDelta: '{"text":"x"}' },
  602. { type: 'block-end', index: 0, block: { type: 'tool-call', id: callId, name: 'echo', arguments: '{"text":"x"}' } },
  603. { type: 'usage', usage: { inputTokens: 10, outputTokens: 5 } },
  604. { type: 'finish', reason: { kind: 'max-tokens' } },
  605. ]])
  606. const ctx = await harness(adapter)
  607. let executions = 0
  608. ctx.tools.register(defineTool({
  609. name: 'echo',
  610. description: '',
  611. parameters: { text: { type: 'string' } },
  612. async execute() {
  613. executions += 1
  614. return [{ type: 'text', text: 'should not run' }]
  615. },
  616. }))
  617. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  618. const reasons: TurnEndReason[] = []
  619. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  620. send(agent, 'go')
  621. await waitForIdle(ctx, agent)
  622. expect(executions).toBe(0)
  623. expect(agent.session.events.some(e => e.type === 'tool/call')).toBe(false)
  624. expect(agent.session.deriveMessages()).toEqual([{ role: 'user', content: [{ type: 'text', text: 'go' }] }])
  625. expect(reasons).toEqual([{ kind: 'max-tokens' }])
  626. // No-data-loss: a max-tokens step whose only content was a dropped tool call
  627. // has EMPTY assistant content, but its usage must still be represented. It
  628. // rides on an (empty-content) assistant/message — there is no standalone
  629. // usage event — and that empty message is skipped by deriveMessages(), so
  630. // the derived history above is NOT corrupted by a spurious assistant turn.
  631. const assistantMessage = agent.session.events.find(e => e.type === 'assistant/message')
  632. expect(assistantMessage?.type === 'assistant/message' && assistantMessage.data).toEqual({
  633. turn: 1, step: 1, content: [], usage: { inputTokens: 10, outputTokens: 5 },
  634. })
  635. })
  636. it('appends no assistant/message for a max-tokens step with empty content and no usage', async () => {
  637. // A max-tokens step truncated to a dropped tool call AND with no usage chunk
  638. // has nothing to record: empty content and no accounting → no assistant/message
  639. // (the empty-content host exists only to carry usage). The turn still ends
  640. // max-tokens.
  641. const callId = CallId('c1')
  642. const adapter = new MockAdapter([[
  643. { type: 'block-start', index: 0, blockType: 'tool-call' },
  644. { type: 'tool-call-delta', index: 0, id: callId, name: 'echo', argumentsDelta: '{"text":"x"}' },
  645. { type: 'block-end', index: 0, block: { type: 'tool-call', id: callId, name: 'echo', arguments: '{"text":"x"}' } },
  646. { type: 'finish', reason: { kind: 'max-tokens' } },
  647. ]])
  648. const ctx = await harness(adapter)
  649. ctx.tools.register(defineTool({
  650. name: 'echo',
  651. description: '',
  652. parameters: { text: { type: 'string' } },
  653. async execute() { return [{ type: 'text', text: 'should not run' }] },
  654. }))
  655. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  656. const reasons: TurnEndReason[] = []
  657. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  658. send(agent, 'go')
  659. await waitForIdle(ctx, agent)
  660. expect(reasons).toEqual([{ kind: 'max-tokens' }])
  661. expect(agent.session.events.some(e => e.type === 'assistant/message')).toBe(false)
  662. expect(agent.session.deriveMessages()).toEqual([{ role: 'user', content: [{ type: 'text', text: 'go' }] }])
  663. })
  664. it('appends no assistant/message for a normal stop finish with empty content and no usage', async () => {
  665. // A clean `stop` finish that streamed nothing assembled (no blocks) and
  666. // carried no usage chunk has nothing to record: the content-or-usage guard
  667. // on the normal step path suppresses a pure trace-only empty assistant/message.
  668. const adapter = new MockAdapter([[{ type: 'finish', reason: { kind: 'stop' } }]])
  669. const ctx = await harness(adapter)
  670. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  671. const reasons: TurnEndReason[] = []
  672. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  673. send(agent, 'go')
  674. await waitForIdle(ctx, agent)
  675. expect(reasons).toEqual([{ kind: 'completed' }])
  676. expect(agent.session.events.some(e => e.type === 'assistant/message')).toBe(false)
  677. expect(agent.session.deriveMessages()).toEqual([{ role: 'user', content: [{ type: 'text', text: 'go' }] }])
  678. })
  679. it('keeps safe max-tokens assistant content while dropping truncated tool calls', async () => {
  680. const callId = CallId('c1')
  681. const adapter = new MockAdapter([[
  682. { type: 'block-start', index: 0, blockType: 'text' },
  683. { type: 'text-delta', index: 0, text: 'partial text' },
  684. { type: 'block-end', index: 0, block: { type: 'text', text: 'partial text' } },
  685. { type: 'block-start', index: 1, blockType: 'tool-call' },
  686. { type: 'tool-call-delta', index: 1, id: callId, name: 'echo', argumentsDelta: '{"text"' },
  687. { type: 'finish', reason: { kind: 'max-tokens' } },
  688. ]])
  689. const ctx = await harness(adapter)
  690. let stepResults = 0
  691. ctx.on('agent/step-result', async (_agent, _turn, _step, message, next) => {
  692. stepResults += 1
  693. expect(message.content).toEqual([{ type: 'text', text: 'partial text' }])
  694. return next()
  695. })
  696. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  697. send(agent, 'go')
  698. await waitForIdle(ctx, agent)
  699. expect(stepResults).toBe(1)
  700. expect(agent.session.events.some(e => e.type === 'tool/call')).toBe(false)
  701. expect(agent.session.deriveMessages()).toEqual([
  702. { role: 'user', content: [{ type: 'text', text: 'go' }] },
  703. { role: 'assistant', content: [{ type: 'text', text: 'partial text' }] },
  704. ])
  705. })
  706. it('contains a step/end observer failure without changing continuation', async () => {
  707. const adapter = new MockAdapter([
  708. toolCallResponse('c1', 'echo', { text: 'x' }),
  709. textResponse('continued after tool call'),
  710. ])
  711. const ctx = await harness(adapter)
  712. ctx.tools.register(defineTool({
  713. name: 'echo',
  714. description: '',
  715. parameters: { text: { type: 'string' } },
  716. async execute(args) {
  717. return [{ type: 'text', text: String(args.text) }]
  718. },
  719. }))
  720. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  721. let threw = false
  722. // Post-commit session observers cannot control the loop. The tool call still
  723. // drives the second model request, and the turn completes normally.
  724. ctx.on('session/event', (_session, event) => {
  725. if (event.type === 'step/end' && !threw) { threw = true; throw new Error('bad step/end listener') }
  726. })
  727. send(agent, 'go')
  728. await waitForIdle(ctx, agent)
  729. expect(adapter.requests).toHaveLength(2)
  730. const turnEnd = agent.session.events.findLast(e => e.type === 'turn/end')
  731. expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason.kind).toBe('completed')
  732. })
  733. it('chains queued messages into consecutive turns', async () => {
  734. const adapter = new MockAdapter([textResponse('first'), textResponse('second')])
  735. const ctx = await harness(adapter)
  736. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  737. const turns: number[] = []
  738. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/start') turns.push(event.data.turn) })
  739. // queue two messages while idle — first starts turn 1 immediately;
  740. // queue the second during turn 1 when the first assistant chunk streams
  741. let queued = false
  742. ctx.on('session/event', (_s, event) => {
  743. if (event.type === 'assistant/chunk' && !queued) {
  744. queued = true
  745. send(agent, 'second message')
  746. }
  747. })
  748. send(agent, 'first message')
  749. await waitForIdle(ctx, agent)
  750. expect(turns).toEqual([1, 2])
  751. expect(adapter.requests).toHaveLength(2)
  752. })
  753. it('awaits session/flush at turn end (persistence checkpoint)', async () => {
  754. const adapter = new MockAdapter([textResponse('ok')])
  755. const ctx = await harness(adapter)
  756. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  757. let flushed = 0
  758. let flushedBeforeIdle = false
  759. ctx.on('session/flush', async (session) => {
  760. await new Promise(r => setTimeout(r, 10))
  761. flushed++
  762. flushedBeforeIdle = agent.status !== 'idle'
  763. void session
  764. })
  765. send(agent, 'hi')
  766. await waitForIdle(ctx, agent)
  767. expect(flushed).toBe(1)
  768. expect(flushedBeforeIdle).toBe(true)
  769. })
  770. it('errors from the model surface as agent/error and end the turn', async () => {
  771. const adapter = new MockAdapter([]) // script exhausted → throws
  772. const ctx = await harness(adapter)
  773. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  774. const errors: Error[] = []
  775. const reasons: TurnEndReason[] = []
  776. ctx.on('agent/error', (_agent, _turn, _step, error) => void errors.push(error))
  777. ctx.on('session/event', (_s, event) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  778. send(agent, 'hi')
  779. await waitForIdle(ctx, agent)
  780. expect(errors).toHaveLength(1)
  781. expect(errors[0]!.message).toContain('script exhausted')
  782. expect(reasons[0]).toMatchObject({ kind: 'error' })
  783. // The durable failure lives entirely on turn/end.reason (with the failing
  784. // step), not a standalone error event.
  785. const turnEnd = agent.session.events.find(e => e.type === 'turn/end')
  786. expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toMatchObject({ kind: 'error', step: 1 })
  787. })
  788. it('disposing the loop fiber mid-turn stops the loop (HMR safety)', async () => {
  789. const adapter = new MockAdapter(['hang'])
  790. const ctx = await harness(adapter)
  791. let agent!: Agent
  792. const fiber = await ctx.plugin(Object.assign((inner: Context) => {
  793. agent = inner.agentLoop.create(SessionId('scoped'), { model: 'mock' })
  794. }, { inject: ['agentLoop'] }))
  795. expect(ctx.agents.get(SessionId('scoped'))).toBe(agent)
  796. send(agent, 'go')
  797. await new Promise(r => setTimeout(r, 30))
  798. expect(agent.status).toBe('running')
  799. await fiber.dispose()
  800. await driverDone(agent)
  801. expect(agent.status).toBe('disposed')
  802. expect(ctx.agents.get(SessionId('scoped'))).toBeUndefined()
  803. expect(() => { send(agent, 'too late') }).toThrow('disposed')
  804. })
  805. it('creates agents from config on startup', async () => {
  806. const adapter = new MockAdapter([textResponse('from config')])
  807. const ctx = new Context()
  808. await ctx.plugin(LlmService)
  809. await ctx.plugin(SessionStore)
  810. await ctx.plugin(SystemPrompt)
  811. await ctx.plugin(ToolRegistry)
  812. await ctx.plugin(AgentRegistry)
  813. await ctx.plugin(AgentLoop, {
  814. agents: [{ id: 'config-agent', model: 'mock' }],
  815. })
  816. ctx.llm.registerAdapter(['mock'], adapter)
  817. const agent = ctx.agents.list()[0]!
  818. expect(agent).toBeDefined()
  819. expect(agent.id).toBe(agent.session.id)
  820. expect(agent.id).toMatch(/^config-agent-session-/)
  821. expect(agent.options.model).toBe('mock')
  822. // the agent is alive: send triggers a turn
  823. send(agent, 'hi')
  824. await waitForIdle(ctx, agent)
  825. expect(adapter.requests).toHaveLength(1)
  826. })
  827. it('attaches config agent cwd to the fresh session header', async () => {
  828. const ctx = new Context()
  829. await ctx.plugin(LlmService)
  830. await ctx.plugin(SessionStore)
  831. await ctx.plugin(SystemPrompt)
  832. await ctx.plugin(ToolRegistry)
  833. await ctx.plugin(AgentRegistry)
  834. await ctx.plugin(AgentLoop, {
  835. agents: [{ id: 'config-agent', model: 'mock', cwd: '/work/project' }],
  836. })
  837. const agent = ctx.agents.list()[0]!
  838. expect(agent.session.header.cwd).toBe('/work/project')
  839. })
  840. it('replays a session log into an identical derived history', async () => {
  841. const adapter = new MockAdapter([
  842. toolCallResponse('c1', 'echo', { text: 'x' }),
  843. textResponse('done'),
  844. ])
  845. const ctx = await harness(adapter)
  846. ctx.tools.register(defineTool({
  847. name: 'echo',
  848. description: '',
  849. parameters: { text: { type: 'string' } },
  850. async execute(args) {
  851. return [{ type: 'text', text: String(args.text) }]
  852. },
  853. }))
  854. const agent = ctx.agentLoop.create(SessionId('a1'), { model: 'mock' })
  855. send(agent, 'run')
  856. await waitForIdle(ctx, agent)
  857. const replayed = ctx.sessions.create(SessionId('replayed'), { seed: [...agent.session.events] })
  858. expect(replayed.deriveMessages()).toEqual(agent.session.deriveMessages())
  859. // event-by-event identity of types
  860. expect(replayed.events.map(e => e.type)).toEqual(
  861. agent.session.events.map(e => e.type))
  862. })
  863. })