interception.spec.ts 33 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767
  1. import { describe, expect, it, vi } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { createUserMessage, CallId } from '@deepseek-ai/dsh-llm'
  4. import SessionStore, {
  5. SessionId,
  6. type SessionEvent,
  7. type TurnEndReason,
  8. type UserMessage,
  9. } from '@deepseek-ai/dsh-session'
  10. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  11. import ToolRegistry, { defineContentToolFixture, type PostToolDecision, type PreToolDecision } from '@deepseek-ai/dsh-tools'
  12. import AgentRegistry, {
  13. type Agent,
  14. type PromptDecision,
  15. type SessionStartSource,
  16. } from '@deepseek-ai/dsh-agent'
  17. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  18. import { MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts'
  19. /**
  20. * The interception seams introduced by the hooks taxonomy: `agent/prompt-submit`,
  21. * `agent/session-start`, `agent/turn-stopping`, and the
  22. * `tools/pre-execute` / `tools/post-execute`
  23. * split with `additionalContexts` buffering. These verify the canonical event
  24. * surface a hook bridge (or a native plugin) programs against, WITHOUT any
  25. * external protocol — a native plugin uses the typed decisions directly.
  26. */
  27. async function harness(adapter: MockAdapter) {
  28. const ctx = new Context()
  29. await ctx.plugin(LlmService)
  30. await ctx.plugin(SessionStore)
  31. await ctx.plugin(SystemPrompt)
  32. await ctx.plugin(ToolRegistry)
  33. await ctx.plugin(AgentRegistry)
  34. await ctx.plugin(AgentLoop, { agents: [] })
  35. ctx.llm.registerAdapter(['mock'], adapter)
  36. return ctx
  37. }
  38. function waitForIdle(ctx: Context, agent: Agent): Promise<void> {
  39. return new Promise((resolve) => {
  40. const dispose = ctx.on('agent/status', (subject, status) => {
  41. if (subject === agent && status === 'idle') {
  42. dispose()
  43. resolve()
  44. }
  45. })
  46. })
  47. }
  48. function send(agent: Agent, text: string) {
  49. agent.followup(createUserMessage({ content: [{ type: 'text', text }], source: { kind: 'user' } }))
  50. }
  51. function events(agent: Agent): SessionEvent[] {
  52. return [...agent.session.events]
  53. }
  54. describe('agent/prompt-submit', () => {
  55. it('allow (default via next) records the user/message unchanged', async () => {
  56. const adapter = new MockAdapter([textResponse('ok')])
  57. const ctx = await harness(adapter)
  58. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  59. const seen: string[] = []
  60. ctx.on('agent/prompt-submit', async (_agent, messages, _signal, next) => {
  61. seen.push(messages[0]!.content.map(b => (b.type === 'text' ? b.text : '')).join(''))
  62. return next()
  63. })
  64. send(agent, 'hello')
  65. await waitForIdle(ctx, agent)
  66. expect(seen).toEqual(['hello'])
  67. const userMsg = events(agent).find(e => e.type === 'user/message')
  68. expect(userMsg?.type === 'user/message' && userMsg.data.content).toEqual([{ type: 'text', text: 'hello' }])
  69. })
  70. it('publishes frozen input without replacing its identity', async () => {
  71. const adapter = new MockAdapter([textResponse('ok')])
  72. const ctx = await harness(adapter)
  73. const agent = ctx.agentLoop.create(SessionId('owned-input'), { provider: 'mock', model: 'mock' })
  74. const entered = Promise.withResolvers<undefined>()
  75. const decision = Promise.withResolvers<PromptDecision>()
  76. const observed: UserMessage[] = []
  77. ctx.on('agent/prompt-submit', async (subject, messages) => {
  78. if (subject !== agent) return { kind: 'allow', messages }
  79. const message = messages[0]!
  80. expect(Object.isFrozen(message)).toBe(true)
  81. expect(Object.isFrozen(message.content)).toBe(true)
  82. expect(Object.isFrozen(message.content[0])).toBe(true)
  83. expect(Object.isFrozen(message.source)).toBe(true)
  84. expect(() => {
  85. const block = message.content[0]
  86. if (block?.type === 'text') block.text = 'listener mutation'
  87. }).toThrow()
  88. observed.push(message)
  89. entered.resolve(undefined)
  90. return decision.promise
  91. })
  92. const input: UserMessage = createUserMessage({
  93. content: [{ type: 'text', text: 'accepted text' }],
  94. source: { kind: 'plugin', plugin: 'accepted source' },
  95. })
  96. const idle = waitForIdle(ctx, agent)
  97. agent.followup(input)
  98. await entered.promise
  99. const block = input.content[0]
  100. expect(() => {
  101. if (block?.type === 'text') block.text = 'caller mutation'
  102. }).toThrow(TypeError)
  103. expect(() => {
  104. if (input.source.kind === 'plugin') input.source.plugin = 'caller mutation'
  105. }).toThrow(TypeError)
  106. decision.resolve({ kind: 'allow', messages: [input] })
  107. await idle
  108. expect(observed).toHaveLength(1)
  109. expect(observed[0]).toBe(input)
  110. expect(observed[0]).toMatchObject({
  111. content: [{ type: 'text', text: 'accepted text' }],
  112. source: { kind: 'plugin', plugin: 'accepted source' },
  113. })
  114. const userMsg = events(agent).find(event => event.type === 'user/message')
  115. expect(userMsg?.type === 'user/message' && userMsg.data).toEqual(input)
  116. })
  117. it('allow with content REWRITES the prompt before it is recorded', async () => {
  118. const adapter = new MockAdapter([textResponse('ok')])
  119. const ctx = await harness(adapter)
  120. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  121. ctx.on('agent/prompt-submit', async (_agent, messages): Promise<PromptDecision> =>
  122. ({
  123. kind: 'allow',
  124. messages: [{ ...messages[0]!, content: [{ type: 'text', text: 'REWRITTEN' }] }],
  125. }))
  126. send(agent, 'original')
  127. await waitForIdle(ctx, agent)
  128. const userMsg = events(agent).find(e => e.type === 'user/message')
  129. expect(userMsg?.type === 'user/message' && userMsg.data.content).toEqual([{ type: 'text', text: 'REWRITTEN' }])
  130. // the rewritten prompt is what reached the model
  131. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('REWRITTEN')
  132. expect(JSON.stringify(adapter.requests[0]!.messages)).not.toContain('original')
  133. })
  134. it('allow with additionalContexts injects separate injected-context user messages into the turn', async () => {
  135. const adapter = new MockAdapter([textResponse('ok')])
  136. const ctx = await harness(adapter)
  137. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  138. ctx.on('agent/prompt-submit', async (_agent, messages): Promise<PromptDecision> =>
  139. ({
  140. kind: 'allow',
  141. messages: [...messages, createUserMessage({
  142. content: [{ type: 'text', text: '<system-reminder>extra ctx</system-reminder>' }],
  143. source: { kind: 'plugin', plugin: 'test' },
  144. })],
  145. }))
  146. send(agent, 'go')
  147. await waitForIdle(ctx, agent)
  148. const log = events(agent)
  149. const userMsg = log.find(e => e.type === 'user/message' && e.data.source.kind === 'user')
  150. const ctxMsg = log.find(e => e.type === 'user/message' && e.data.source.kind === 'plugin')
  151. expect(userMsg).toBeDefined()
  152. expect(ctxMsg?.type === 'user/message' && ctxMsg.data.content).toEqual([{ type: 'text', text: '<system-reminder>extra ctx</system-reminder>' }])
  153. expect(ctxMsg?.type === 'user/message' && ctxMsg.data.source).toEqual({ kind: 'plugin', plugin: 'test' })
  154. const sent = JSON.stringify(adapter.requests[0]!.messages)
  155. expect(sent).toContain('extra ctx')
  156. })
  157. it('runs pre-step after prompt rewrites and injected context become durable', async () => {
  158. const adapter = new MockAdapter([textResponse('ok')])
  159. const ctx = await harness(adapter)
  160. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  161. ctx.on('agent/prompt-submit', async (_agent, messages): Promise<PromptDecision> =>
  162. ({
  163. kind: 'allow',
  164. messages: [{
  165. ...messages[0]!,
  166. content: [{ type: 'text', text: 'REWRITTEN prompt' }],
  167. }, createUserMessage({
  168. content: [{ type: 'text', text: 'injected ctx' }], source: { kind: 'plugin', plugin: 'test' },
  169. })],
  170. }))
  171. let preStepDerived: string | undefined
  172. ctx.on('agent/step', (subject, _turn, step) => {
  173. if (subject === agent && step === 1) preStepDerived = JSON.stringify(subject.session.deriveMessages())
  174. })
  175. send(agent, 'ORIGINAL prompt')
  176. await waitForIdle(ctx, agent)
  177. expect(preStepDerived).toBeDefined()
  178. expect(preStepDerived).toContain('REWRITTEN prompt')
  179. expect(preStepDerived).toContain('injected ctx')
  180. expect(preStepDerived).not.toContain('ORIGINAL prompt')
  181. })
  182. it('block drops the claimed prompt before any turn or model call', async () => {
  183. const adapter = new MockAdapter([textResponse('should not run')])
  184. const ctx = await harness(adapter)
  185. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  186. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  187. ({ kind: 'block', reason: 'blocked by policy' }))
  188. const reasons: TurnEndReason[] = []
  189. ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  190. agent.followup(createUserMessage({ content: [{ type: 'text', text: 'do something' }], source: { kind: 'user' } }))
  191. await agent.whenIdle()
  192. // the model was never called
  193. expect(adapter.requests).toHaveLength(0)
  194. const log = events(agent)
  195. expect(log.some(e => e.type === 'turn/start')).toBe(false)
  196. expect(log.some(e => e.type === 'turn/end')).toBe(false)
  197. expect(log.some(e => e.type === 'user/message')).toBe(false)
  198. expect(log.some(e => e.type === 'step/start')).toBe(false)
  199. expect(reasons).toEqual([])
  200. })
  201. it('stages inject and steer during admission for the admitted turn', async () => {
  202. const adapter = new MockAdapter([textResponse('ok')])
  203. const ctx = await harness(adapter)
  204. const agent = ctx.agentLoop.create(SessionId('admission-outbox'), { provider: 'mock', model: 'mock' })
  205. const entered = Promise.withResolvers<undefined>()
  206. const decision = Promise.withResolvers<PromptDecision>()
  207. let claimed: UserMessage[] = []
  208. ctx.on('agent/prompt-submit', async (_agent, messages) => {
  209. claimed = messages
  210. entered.resolve(undefined)
  211. return decision.promise
  212. })
  213. const idle = waitForIdle(ctx, agent)
  214. send(agent, 'admitted prompt')
  215. await entered.promise
  216. expect(agent.status).toBe('running')
  217. expect(events(agent).some(event => event.type === 'turn/start')).toBe(false)
  218. agent.inject(createUserMessage({
  219. content: [{ type: 'text', text: 'attached context' }],
  220. source: { kind: 'plugin', plugin: 'test' },
  221. }))
  222. agent.steer(createUserMessage({ content: [{ type: 'text', text: 'admission steering' }], source: { kind: 'user' } }))
  223. expect(events(agent).some(event => event.type === 'user/message')).toBe(false)
  224. expect(agent.inbox.nextStep.map(message => message.content[0]))
  225. .toEqual([
  226. { type: 'text', text: 'attached context' },
  227. { type: 'text', text: 'admission steering' },
  228. ])
  229. decision.resolve({ kind: 'allow', messages: claimed })
  230. await idle
  231. expect(agent.inbox.hasPending).toBe(false)
  232. const staged = events(agent).filter(event =>
  233. event.type === 'turn/start' || event.type === 'user/message' || event.type === 'steering/message')
  234. expect(staged.map(event => event.type)).toEqual([
  235. 'turn/start',
  236. 'user/message',
  237. 'user/message',
  238. 'steering/message',
  239. ])
  240. expect(staged[1]?.type === 'user/message' && staged[1].data.content)
  241. .toEqual([{ type: 'text', text: 'admitted prompt' }])
  242. expect(staged[2]?.type === 'user/message' && staged[2].data.content)
  243. .toEqual([{ type: 'text', text: 'attached context' }])
  244. expect(staged[3]?.type === 'steering/message' && staged[3].data.message.content)
  245. .toEqual([{ type: 'text', text: 'admission steering' }])
  246. const request = JSON.stringify(adapter.requests[0]?.messages)
  247. expect(request).toContain('admitted prompt')
  248. expect(request).toContain('attached context')
  249. expect(request).toContain('admission steering')
  250. })
  251. it('keeps admission-time outbox input staged when admission is blocked', async () => {
  252. const adapter = new MockAdapter([textResponse('retried')])
  253. const ctx = await harness(adapter)
  254. const agent = ctx.agentLoop.create(SessionId('blocked-admission-outbox'), { provider: 'mock', model: 'mock' })
  255. const entered = Promise.withResolvers<undefined>()
  256. const decision = Promise.withResolvers<PromptDecision>()
  257. const disposeBlock = ctx.on('agent/prompt-submit', async () => {
  258. entered.resolve(undefined)
  259. return decision.promise
  260. })
  261. const blockedIdle = waitForIdle(ctx, agent)
  262. send(agent, 'blocked prompt')
  263. await entered.promise
  264. agent.inject(createUserMessage({
  265. content: [{ type: 'text', text: 'staged context' }],
  266. source: { kind: 'plugin', plugin: 'test' },
  267. }))
  268. agent.steer(createUserMessage({ content: [{ type: 'text', text: 'staged steering' }], source: { kind: 'user' } }))
  269. decision.resolve({ kind: 'block', reason: 'policy' })
  270. await blockedIdle
  271. expect(agent.inbox.nextStep).toHaveLength(2)
  272. expect(events(agent)).toEqual([])
  273. expect(adapter.requests).toEqual([])
  274. disposeBlock()
  275. send(agent, 'resume')
  276. await waitForIdle(ctx, agent)
  277. const staged = events(agent).filter(event =>
  278. event.type === 'user/message' || event.type === 'steering/message')
  279. expect(staged.map(event => event.type)).toEqual([
  280. 'user/message',
  281. 'steering/message',
  282. 'user/message',
  283. ])
  284. expect(JSON.stringify(adapter.requests[0]?.messages)).not.toContain('blocked prompt')
  285. expect(JSON.stringify(adapter.requests[0]?.messages)).toContain('staged context')
  286. expect(JSON.stringify(adapter.requests[0]?.messages)).toContain('staged steering')
  287. })
  288. it('orders rejected-admission outbox input before a later admitted prompt', async () => {
  289. const adapter = new MockAdapter([textResponse('continued')])
  290. const ctx = await harness(adapter)
  291. const agent = ctx.agentLoop.create(SessionId('rejected-admission-order'), {
  292. provider: 'mock',
  293. model: 'mock',
  294. })
  295. ctx.on('agent/prompt-submit', async (_agent, messages, _signal, next) => {
  296. const decision = await next()
  297. return messages.some(message =>
  298. message.content.some(block => block.type === 'text' && block.text === 'blocked prompt'))
  299. ? { kind: 'block', reason: 'policy' }
  300. : decision
  301. })
  302. ctx.on('agent/prompt-submit', async (subject, messages, _signal, next) => {
  303. if (messages.some(message =>
  304. message.content.some(block => block.type === 'text' && block.text === 'blocked prompt'))) {
  305. subject.inject(createUserMessage({
  306. content: [{ type: 'text', text: 'earlier state change' }],
  307. source: { kind: 'plugin', plugin: 'test' },
  308. }))
  309. subject.steer(createUserMessage({
  310. content: [{ type: 'text', text: 'earlier steering' }],
  311. source: { kind: 'user' },
  312. }))
  313. }
  314. return next()
  315. })
  316. const idle = waitForIdle(ctx, agent)
  317. send(agent, 'blocked prompt')
  318. send(agent, 'later prompt')
  319. await idle
  320. const staged = events(agent).filter(event =>
  321. event.type === 'turn/start' || event.type === 'user/message' || event.type === 'steering/message')
  322. expect(staged.map(event => event.type)).toEqual([
  323. 'turn/start',
  324. 'user/message',
  325. 'steering/message',
  326. 'user/message',
  327. ])
  328. expect(staged[1]?.type === 'user/message' && staged[1].data.content)
  329. .toEqual([{ type: 'text', text: 'earlier state change' }])
  330. expect(staged[2]?.type === 'steering/message' && staged[2].data.message.content)
  331. .toEqual([{ type: 'text', text: 'earlier steering' }])
  332. expect(staged[3]?.type === 'user/message' && staged[3].data.content)
  333. .toEqual([{ type: 'text', text: 'later prompt' }])
  334. })
  335. it('commits context-only injection when admission closes without a turn', async () => {
  336. const adapter = new MockAdapter([])
  337. const ctx = await harness(adapter)
  338. const agent = ctx.agentLoop.create(SessionId('blocked-admission-context'), { provider: 'mock', model: 'mock' })
  339. const entered = Promise.withResolvers<undefined>()
  340. const decision = Promise.withResolvers<PromptDecision>()
  341. ctx.on('agent/prompt-submit', async () => {
  342. entered.resolve(undefined)
  343. return decision.promise
  344. })
  345. const idle = waitForIdle(ctx, agent)
  346. send(agent, 'blocked prompt')
  347. await entered.promise
  348. agent.inject(createUserMessage({
  349. content: [{ type: 'text', text: 'independent context' }],
  350. source: { kind: 'plugin', plugin: 'test' },
  351. }))
  352. decision.resolve({ kind: 'block', reason: 'policy' })
  353. await idle
  354. const log = events(agent)
  355. expect(log.map(event => event.type)).toEqual(['user/message'])
  356. expect(log[0]?.type === 'user/message' && log[0].data.content)
  357. .toEqual([{ type: 'text', text: 'independent context' }])
  358. expect(adapter.requests).toEqual([])
  359. })
  360. it('retains rejected-admission context when its idle append fails', async () => {
  361. const adapter = new MockAdapter([textResponse('retried')])
  362. const ctx = await harness(adapter)
  363. const agent = ctx.agentLoop.create(SessionId('blocked-admission-append-failure'), {
  364. provider: 'mock',
  365. model: 'mock',
  366. })
  367. const warned = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined)
  368. vi.spyOn(agent.session, 'append').mockImplementationOnce(() => {
  369. throw new Error('append unavailable')
  370. })
  371. const entered = Promise.withResolvers<undefined>()
  372. const decision = Promise.withResolvers<PromptDecision>()
  373. const disposeBlock = ctx.on('agent/prompt-submit', async () => {
  374. entered.resolve(undefined)
  375. return decision.promise
  376. })
  377. agent.followup(createUserMessage({ content: [{ type: 'text', text: 'blocked prompt' }], source: { kind: 'user' } }))
  378. await entered.promise
  379. agent.inject(createUserMessage({
  380. content: [{ type: 'text', text: 'retained context' }],
  381. source: { kind: 'plugin', plugin: 'test' },
  382. }))
  383. decision.resolve({ kind: 'block', reason: 'policy' })
  384. await agent.whenIdle()
  385. expect(events(agent)).toEqual([])
  386. expect(warned).toHaveBeenCalledWith(expect.stringContaining('append unavailable'))
  387. disposeBlock()
  388. send(agent, 'resume')
  389. await waitForIdle(ctx, agent)
  390. expect(events(agent).some(event => event.type === 'user/message'
  391. && JSON.stringify(event.data.content).includes('retained context'))).toBe(true)
  392. })
  393. it('adjacent blocked and allowed prompts keep independent turn outcomes', async () => {
  394. const adapter = new MockAdapter([textResponse('ran once')])
  395. const ctx = await harness(adapter)
  396. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  397. ctx.on('agent/prompt-submit', async (_agent, messages, _signal, next): Promise<PromptDecision> => {
  398. const text = messages.flatMap(message => message.content)
  399. .map(b => (b.type === 'text' ? b.text : '')).join('')
  400. return text === 'secret' ? { kind: 'block', reason: 'policy: no secrets' } : next()
  401. })
  402. const reasons: TurnEndReason[] = []
  403. ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  404. // The rejected admission is dropped; the allowed prompt owns the only turn.
  405. send(agent, 'secret')
  406. send(agent, 'safe')
  407. await waitForIdle(ctx, agent)
  408. const log = events(agent)
  409. // The allowed prompt became a user/message and drove exactly one model call.
  410. const userMsgs = log.filter(e => e.type === 'user/message')
  411. expect(userMsgs).toHaveLength(1)
  412. expect(userMsgs[0]?.type === 'user/message' && userMsgs[0].data.content).toEqual([{ type: 'text', text: 'safe' }])
  413. expect(adapter.requests.length).toBeGreaterThanOrEqual(1)
  414. expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1)
  415. expect(reasons).toEqual([{ kind: 'completed' }])
  416. })
  417. it('a throwing prompt-submit listener drops that admission while an adjacent message survives', async () => {
  418. const adapter = new MockAdapter([textResponse('after')])
  419. const ctx = await harness(adapter)
  420. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  421. let threw = false
  422. ctx.on('agent/prompt-submit', async (_agent, messages) => {
  423. if (!threw) { threw = true; throw new Error('prompt hook broke') }
  424. return { kind: 'allow' as const, messages }
  425. })
  426. const errors: Error[] = []
  427. const reasons: TurnEndReason[] = []
  428. const statuses: string[] = []
  429. ctx.on('agent/error', (_a, _t, _s, error) => {
  430. if (error instanceof Error) errors.push(error)
  431. })
  432. ctx.on('agent/status', (subject, status) => { if (subject === agent) statuses.push(status) })
  433. ctx.on('session/event', (session, event) => {
  434. if (session === agent.session && event.type === 'turn/end') reasons.push(event.data.reason)
  435. })
  436. const idle = waitForIdle(ctx, agent)
  437. send(agent, 'first')
  438. send(agent, 'second')
  439. await idle
  440. expect(errors).toEqual([])
  441. const log = events(agent)
  442. expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1)
  443. expect(log.filter(e => e.type === 'turn/end')).toHaveLength(1)
  444. expect(reasons).toEqual([{ kind: 'completed' }])
  445. expect(statuses).toEqual(['running', 'idle'])
  446. expect(adapter.requests).toHaveLength(1)
  447. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('second')
  448. })
  449. })
  450. describe('agent/session-start', () => {
  451. it('fires once with source "startup" for a fresh create, before the first turn', async () => {
  452. const adapter = new MockAdapter([textResponse('ok')])
  453. const ctx = await harness(adapter)
  454. const sources: SessionStartSource[] = []
  455. ctx.on('agent/session-start', (_agent, source) => void sources.push(source))
  456. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  457. // fires synchronously at create, before any turn
  458. expect(sources).toEqual(['startup'])
  459. expect(events(agent).some(e => e.type === 'turn/start')).toBe(false)
  460. send(agent, 'go')
  461. await waitForIdle(ctx, agent)
  462. // still only one session-start
  463. expect(sources).toEqual(['startup'])
  464. })
  465. it('a session-start listener can inject context the first request sees', async () => {
  466. const adapter = new MockAdapter([textResponse('ok')])
  467. const ctx = await harness(adapter)
  468. ctx.on('agent/session-start', (agent) => {
  469. agent.inject(createUserMessage({ content: [{ type: 'text', text: 'session preamble' }], source: { kind: 'plugin', plugin: 'test' } }))
  470. })
  471. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  472. send(agent, 'go')
  473. await waitForIdle(ctx, agent)
  474. // the injected context reached the model on the first (only) request
  475. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('session preamble')
  476. // and is recorded with the plugin source, never mislabeled as a user prompt
  477. const ctxMsg = events(agent).find(e => e.type === 'user/message' && e.data.source.kind === 'plugin')
  478. expect(ctxMsg?.type === 'user/message' && ctxMsg.data.source).toEqual({ kind: 'plugin', plugin: 'test' })
  479. })
  480. it('a throwing session-start listener does not abort agent construction', async () => {
  481. const adapter = new MockAdapter([textResponse('ok')])
  482. const ctx = await harness(adapter)
  483. ctx.on('agent/session-start', () => { throw new Error('session-start hook broke') })
  484. // create must not throw — the listener error is contained/logged
  485. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  486. expect(agent.id).toBe(SessionId('a1'))
  487. // and the agent still runs
  488. send(agent, 'go')
  489. await waitForIdle(ctx, agent)
  490. expect(adapter.requests).toHaveLength(1)
  491. })
  492. })
  493. describe('tool additionalContexts buffering across a step', () => {
  494. it('appends each call\'s contexts only AFTER all tool/results, preserving adjacency', async () => {
  495. // One assistant step with TWO tool calls; the second model response stops.
  496. const twoCalls = [
  497. { type: 'block-start' as const, index: 0, blockType: 'tool-call' as const },
  498. { type: 'block-end' as const, index: 0, block: { type: 'tool-call' as const, id: CallId('c1'), name: 'echo', arguments: '{"text":"a"}' } },
  499. { type: 'block-start' as const, index: 1, blockType: 'tool-call' as const },
  500. { type: 'block-end' as const, index: 1, block: { type: 'tool-call' as const, id: CallId('c2'), name: 'echo', arguments: '{"text":"b"}' } },
  501. { type: 'usage' as const, usage: { inputTokens: 5, outputTokens: 5 } },
  502. { type: 'finish' as const, reason: { kind: 'tool-calls' as const } },
  503. ]
  504. const adapter = new MockAdapter([twoCalls, textResponse('done')])
  505. const ctx = await harness(adapter)
  506. ctx.tools.register(defineContentToolFixture({
  507. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  508. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  509. }))
  510. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  511. // Each call attaches one context naming itself.
  512. ctx.on('tools/post-execute', async (exec, _result): Promise<PostToolDecision> =>
  513. ({
  514. kind: 'accept',
  515. additionalContexts: [createUserMessage({
  516. content: [{ type: 'text', text: `ctx-${exec.callId}` }],
  517. source: { kind: 'plugin', plugin: 'p' },
  518. })],
  519. }))
  520. send(agent, 'go')
  521. await waitForIdle(ctx, agent)
  522. // Event order in the log: both tool/results, THEN both injected contexts —
  523. // never interleaved (which would break tool-call/result adjacency).
  524. const injected = events(agent).filter(e => e.type === 'user/message' && e.data.source.kind === 'plugin')
  525. const seqs = events(agent)
  526. const firstResult = seqs.findIndex(e => e.type === 'tool/result')
  527. const lastResult = seqs.map(e => e.type).lastIndexOf('tool/result')
  528. const firstCtx = seqs.findIndex(e => e === injected[0])
  529. expect(firstResult).toBeGreaterThanOrEqual(0)
  530. expect(lastResult).toBeGreaterThan(firstResult) // two results
  531. expect(firstCtx).toBeGreaterThan(lastResult) // context only after ALL results
  532. // both contexts present
  533. const ctxTexts = injected
  534. .flatMap(e => (e.type === 'user/message' ? e.data.content : []))
  535. .map(b => (b.type === 'text' ? b.text : ''))
  536. expect(ctxTexts).toEqual(['ctx-c1', 'ctx-c2'])
  537. })
  538. it('appends multiple contexts deferred by one composite tool after its outer result', async () => {
  539. const adapter = new MockAdapter([toolCallResponse('c1', 'composite', {}), textResponse('done')])
  540. const ctx = await harness(adapter)
  541. ctx.tools.register(defineContentToolFixture({
  542. name: 'composite', description: 'composite', parameters: {},
  543. async execute(_args, exec) {
  544. exec.deferContext(createUserMessage({
  545. content: [{ type: 'text', text: 'nested-a' }], source: { kind: 'plugin', plugin: 'a' },
  546. }))
  547. exec.deferContext(createUserMessage({
  548. content: [{ type: 'text', text: 'nested-b' }], source: { kind: 'plugin', plugin: 'b' },
  549. }))
  550. return [{ type: 'text', text: 'outer result' }]
  551. },
  552. }))
  553. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  554. send(agent, 'go')
  555. await waitForIdle(ctx, agent)
  556. const log = events(agent)
  557. const resultIndex = log.findIndex(event => event.type === 'tool/result')
  558. const contextEvents = log.filter(event => event.type === 'user/message' && event.data.source.kind === 'plugin')
  559. expect(resultIndex).toBeGreaterThanOrEqual(0)
  560. expect(log.findIndex(event => event === contextEvents[0])).toBeGreaterThan(resultIndex)
  561. expect(contextEvents.map(event => event.type === 'user/message' && event.data.source)).toEqual([
  562. { kind: 'plugin', plugin: 'a' },
  563. { kind: 'plugin', plugin: 'b' },
  564. ])
  565. })
  566. })
  567. describe('tools/pre-execute gate (native-plugin permission pattern, end-to-end through the loop)', () => {
  568. it('deny short-circuits dispatch into an isError result the model sees', async () => {
  569. const adapter = new MockAdapter([toolCallResponse('c1', 'danger', {}), textResponse('ok')])
  570. const ctx = await harness(adapter)
  571. let ran = false
  572. ctx.tools.register(defineContentToolFixture({
  573. name: 'danger', description: 'danger', parameters: {},
  574. async execute() { ran = true; return [{ type: 'text', text: 'should not run' }] },
  575. }))
  576. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  577. ctx.on('tools/pre-execute', async (exec, next): Promise<PreToolDecision> => {
  578. if (exec.name === 'danger') return { kind: 'deny', reason: 'blocked dangerous tool' }
  579. return next()
  580. })
  581. send(agent, 'go')
  582. await waitForIdle(ctx, agent)
  583. expect(ran).toBe(false)
  584. const result = events(agent).find(e => e.type === 'tool/result')
  585. expect(result?.type === 'tool/result' && result.data.message.content[0].isError).toBe(true)
  586. expect(result?.type === 'tool/result'
  587. && result.data.message.content[0].content.some(b => b.type === 'text' && b.text.includes('blocked dangerous tool'))).toBe(true)
  588. })
  589. })
  590. describe('worked example: a native hook plugin is just a cordis plugin on the seams', () => {
  591. // The whole point of the interception taxonomy: a "native hook" needs no dsh-hook-protocol,
  592. // no external command, no hook/* log — it is an ordinary cordis plugin subscribing to the
  593. // canonical events and returning typed decisions.
  594. const NativeGuard = {
  595. name: 'native-guard',
  596. apply(ctx: Context) {
  597. // 1. SessionStart: seed a standing instruction.
  598. ctx.on('agent/session-start', (agent, source) => {
  599. agent.inject(createUserMessage({ content: [{ type: 'text', text: `policy active (started: ${source})` }], source: { kind: 'plugin', plugin: 'native-guard' } }))
  600. })
  601. // 2. PromptSubmit: block a forbidden prompt, annotate the rest.
  602. ctx.on('agent/prompt-submit', async (_agent, messages, _signal, next): Promise<PromptDecision> => {
  603. const text = messages.flatMap(message => message.content)
  604. .map(b => (b.type === 'text' ? b.text : '')).join('')
  605. if (text.includes('rm -rf')) return { kind: 'block', reason: 'destructive prompt blocked' }
  606. return next()
  607. })
  608. // 3. PreToolUse: deny a dangerous tool by name.
  609. ctx.on('tools/pre-execute', async (exec, next): Promise<PreToolDecision> => {
  610. if (exec.name === 'danger') return { kind: 'deny', reason: 'danger tool denied' }
  611. return next()
  612. })
  613. // 4. PostToolUse: attach context after a tool runs.
  614. ctx.on('tools/post-execute', async (_exec, _result, next): Promise<PostToolDecision> => {
  615. const decision = await next()
  616. if (decision.kind === 'accept') {
  617. return { kind: 'accept', additionalContexts: [createUserMessage({
  618. content: [{ type: 'text', text: 'audited' }], source: { kind: 'plugin', plugin: 'native-guard' },
  619. })] }
  620. }
  621. return decision
  622. })
  623. },
  624. }
  625. it('all four seams fire for a real allowed turn with a tool call', async () => {
  626. const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'hi' }), textResponse('done')])
  627. const ctx = await harness(adapter)
  628. await ctx.plugin(NativeGuard)
  629. ctx.tools.register(defineContentToolFixture({
  630. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  631. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  632. }))
  633. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  634. send(agent, 'please echo hi')
  635. await waitForIdle(ctx, agent)
  636. const log = events(agent)
  637. // session-start preamble injected
  638. expect(log.some(e => e.type === 'user/message' && e.data.source.kind === 'plugin'
  639. && e.data.content.some(b => b.type === 'text' && b.text.includes('policy active (started: startup)')))).toBe(true)
  640. // prompt allowed → user-sourced user/message recorded
  641. expect(log.some(e => e.type === 'user/message' && e.data.source.kind === 'user')).toBe(true)
  642. // tool ran (echo allowed) and post-execute attached "audited" context
  643. expect(log.some(e => e.type === 'tool/result' && !e.data.message.content[0].isError)).toBe(true)
  644. expect(log.some(e => e.type === 'user/message' && e.data.source.kind === 'plugin'
  645. && e.data.content.some(b => b.type === 'text' && b.text === 'audited'))).toBe(true)
  646. // NO hook/* events — a native plugin needs none
  647. expect(log.some(e => e.type.startsWith('hook/'))).toBe(false)
  648. })
  649. it('the same plugin blocks a destructive prompt before a turn or model call', async () => {
  650. const adapter = new MockAdapter([textResponse('should not run')])
  651. const ctx = await harness(adapter)
  652. await ctx.plugin(NativeGuard)
  653. const agent = ctx.agentLoop.create(SessionId('a2'), { provider: 'mock', model: 'mock' })
  654. const reasons: TurnEndReason[] = []
  655. ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  656. send(agent, 'run rm -rf /')
  657. await agent.whenIdle()
  658. expect(adapter.requests).toHaveLength(0)
  659. expect(reasons).toEqual([])
  660. })
  661. it('HMR-safety: disposing the plugin fiber removes all four listeners', async () => {
  662. const adapter = new MockAdapter([textResponse('ok')])
  663. const ctx = await harness(adapter)
  664. const fiber = await ctx.plugin(NativeGuard)
  665. await fiber.dispose()
  666. // After disposal, a destructive prompt is NOT blocked (the listener is gone).
  667. const agent = ctx.agentLoop.create(SessionId('a3'), { provider: 'mock', model: 'mock' })
  668. send(agent, 'run rm -rf /')
  669. await waitForIdle(ctx, agent)
  670. // the prompt ran (not rejected) — proving the prompt-submit listener was disposed
  671. expect(adapter.requests).toHaveLength(1)
  672. expect(events(agent).some(e => e.type === 'user/message')).toBe(true)
  673. })
  674. })