interception.spec.ts 28 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632
  1. import { describe, expect, it, vi } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { CallId } from '@deepseek-ai/dsh-llm'
  4. import SessionStore, { SessionId, type SessionEvent, type TurnEndReason } from '@deepseek-ai/dsh-session'
  5. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  6. import ToolRegistry, { defineContentToolFixture, type PostToolDecision, type PreToolDecision } from '@deepseek-ai/dsh-tools'
  7. import AgentRegistry, { type Agent, type InboxPlacement, type PromptDecision, type SessionStartSource } from '@deepseek-ai/dsh-agent'
  8. import AgentLoop from '@deepseek-ai/dsh-agent-loop'
  9. import { MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts'
  10. /**
  11. * The interception seams introduced by the hooks taxonomy: `agent/prompt-submit`,
  12. * `agent/session-start`, `agent/turn-stopping`, and the
  13. * `tools/pre-execute` / `tools/post-execute`
  14. * split with `additionalContexts` buffering. These verify the canonical event
  15. * surface a hook bridge (or a native plugin) programs against, WITHOUT any
  16. * external protocol — a native plugin uses the typed decisions directly.
  17. */
  18. async function harness(adapter: MockAdapter) {
  19. const ctx = new Context()
  20. await ctx.plugin(LlmService)
  21. await ctx.plugin(SessionStore)
  22. await ctx.plugin(SystemPrompt)
  23. await ctx.plugin(ToolRegistry)
  24. await ctx.plugin(AgentRegistry)
  25. await ctx.plugin(AgentLoop, { agents: [] })
  26. ctx.llm.registerAdapter(['mock'], adapter)
  27. return ctx
  28. }
  29. function waitForIdle(ctx: Context, agent: Agent): Promise<void> {
  30. return new Promise((resolve) => {
  31. const dispose = ctx.on('agent/status', (subject, status) => {
  32. if (subject === agent && status === 'idle') {
  33. dispose()
  34. resolve()
  35. }
  36. })
  37. })
  38. }
  39. function send(agent: Agent, text: string) {
  40. agent.followup({ content: [{ type: 'text', text }], source: { kind: 'user' } })
  41. }
  42. function events(agent: Agent): SessionEvent[] {
  43. return [...agent.session.events]
  44. }
  45. describe('agent/prompt-submit', () => {
  46. it('allow (default via next) records the user/message unchanged', async () => {
  47. const adapter = new MockAdapter([textResponse('ok')])
  48. const ctx = await harness(adapter)
  49. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  50. const seen: string[] = []
  51. ctx.on('agent/prompt-submit', async (_agent, content, _source, _signal, next) => {
  52. seen.push(content.map(b => (b.type === 'text' ? b.text : '')).join(''))
  53. return next()
  54. })
  55. send(agent, 'hello')
  56. await waitForIdle(ctx, agent)
  57. expect(seen).toEqual(['hello'])
  58. const userMsg = events(agent).find(e => e.type === 'user/message')
  59. expect(userMsg?.type === 'user/message' && userMsg.data.content).toEqual([{ type: 'text', text: 'hello' }])
  60. })
  61. it('allow with content REWRITES the prompt before it is recorded', async () => {
  62. const adapter = new MockAdapter([textResponse('ok')])
  63. const ctx = await harness(adapter)
  64. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  65. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  66. ({ kind: 'allow', content: [{ type: 'text', text: 'REWRITTEN' }] }))
  67. send(agent, 'original')
  68. await waitForIdle(ctx, agent)
  69. const userMsg = events(agent).find(e => e.type === 'user/message')
  70. expect(userMsg?.type === 'user/message' && userMsg.data.content).toEqual([{ type: 'text', text: 'REWRITTEN' }])
  71. // the rewritten prompt is what reached the model
  72. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('REWRITTEN')
  73. expect(JSON.stringify(adapter.requests[0]!.messages)).not.toContain('original')
  74. })
  75. it('allow with additionalContexts injects separate injected-context user messages into the turn', async () => {
  76. const adapter = new MockAdapter([textResponse('ok')])
  77. const ctx = await harness(adapter)
  78. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  79. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  80. ({
  81. kind: 'allow',
  82. additionalContexts: [{
  83. content: [{ type: 'text', text: '<system-reminder>extra ctx</system-reminder>' }],
  84. source: { kind: 'plugin', plugin: 'test' },
  85. }],
  86. }))
  87. send(agent, 'go')
  88. await waitForIdle(ctx, agent)
  89. const log = events(agent)
  90. const userMsg = log.find(e => e.type === 'user/message' && e.data.source.kind === 'user')
  91. const ctxMsg = log.find(e => e.type === 'user/message' && e.data.source.kind === 'plugin')
  92. expect(userMsg).toBeDefined()
  93. expect(ctxMsg?.type === 'user/message' && ctxMsg.data.content).toEqual([{ type: 'text', text: '<system-reminder>extra ctx</system-reminder>' }])
  94. expect(ctxMsg?.type === 'user/message' && ctxMsg.data.source).toEqual({ kind: 'plugin', plugin: 'test' })
  95. const sent = JSON.stringify(adapter.requests[0]!.messages)
  96. expect(sent).toContain('extra ctx')
  97. })
  98. it('runs pre-step after prompt rewrites and injected context become durable', async () => {
  99. const adapter = new MockAdapter([textResponse('ok')])
  100. const ctx = await harness(adapter)
  101. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  102. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  103. ({
  104. kind: 'allow',
  105. content: [{ type: 'text', text: 'REWRITTEN prompt' }],
  106. additionalContexts: [{ content: [{ type: 'text', text: 'injected ctx' }], source: { kind: 'plugin', plugin: 'test' } }],
  107. }))
  108. let preStepDerived: string | undefined
  109. ctx.on('agent/step', (subject, _turn, step) => {
  110. if (subject === agent && step === 1) preStepDerived = JSON.stringify(subject.session.deriveMessages())
  111. })
  112. send(agent, 'ORIGINAL prompt')
  113. await waitForIdle(ctx, agent)
  114. expect(preStepDerived).toBeDefined()
  115. expect(preStepDerived).toContain('REWRITTEN prompt')
  116. expect(preStepDerived).toContain('injected ctx')
  117. expect(preStepDerived).not.toContain('ORIGINAL prompt')
  118. })
  119. it('block drops the claimed prompt before any turn or model call', async () => {
  120. const adapter = new MockAdapter([textResponse('should not run')])
  121. const ctx = await harness(adapter)
  122. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  123. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  124. ({ kind: 'block', reason: 'blocked by policy' }))
  125. const reasons: TurnEndReason[] = []
  126. ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  127. agent.followup({ content: [{ type: 'text', text: 'do something' }], source: { kind: 'user' } })
  128. await agent.whenIdle()
  129. // the model was never called
  130. expect(adapter.requests).toHaveLength(0)
  131. const log = events(agent)
  132. expect(log.some(e => e.type === 'turn/start')).toBe(false)
  133. expect(log.some(e => e.type === 'turn/end')).toBe(false)
  134. expect(log.some(e => e.type === 'user/message')).toBe(false)
  135. expect(log.some(e => e.type === 'step/start')).toBe(false)
  136. expect(reasons).toEqual([])
  137. })
  138. it('stages inject and steer during admission for the admitted turn', async () => {
  139. const adapter = new MockAdapter([textResponse('ok')])
  140. const ctx = await harness(adapter)
  141. const agent = ctx.agentLoop.create(SessionId('admission-outbox'), { provider: 'mock', model: 'mock' })
  142. const entered = Promise.withResolvers<undefined>()
  143. const decision = Promise.withResolvers<PromptDecision>()
  144. const placements: InboxPlacement[] = []
  145. ctx.on('agent/prompt-submit', async () => {
  146. entered.resolve(undefined)
  147. return decision.promise
  148. })
  149. ctx.on('agent/inbox/enqueue', (subject, _message, placement) => {
  150. if (subject === agent) placements.push(placement)
  151. })
  152. const idle = waitForIdle(ctx, agent)
  153. send(agent, 'admitted prompt')
  154. await entered.promise
  155. expect(agent.status).toBe('running')
  156. expect(agent.acceptsNextStep).toBe(true)
  157. expect(events(agent).some(event => event.type === 'turn/start')).toBe(false)
  158. agent.inject({
  159. content: [{ type: 'text', text: 'attached context' }],
  160. source: { kind: 'plugin', plugin: 'test' },
  161. })
  162. agent.steer({ content: [{ type: 'text', text: 'admission steering' }], source: { kind: 'user' } })
  163. expect(events(agent).some(event => event.type === 'user/message')).toBe(false)
  164. expect(placements).toEqual(['queued', 'steering'])
  165. decision.resolve({ kind: 'allow' })
  166. await idle
  167. expect(agent.acceptsNextStep).toBe(false)
  168. const staged = events(agent).filter(event =>
  169. event.type === 'turn/start' || event.type === 'user/message' || event.type === 'steering/message')
  170. expect(staged.map(event => event.type)).toEqual([
  171. 'turn/start',
  172. 'user/message',
  173. 'user/message',
  174. 'steering/message',
  175. ])
  176. expect(staged[1]?.type === 'user/message' && staged[1].data.content)
  177. .toEqual([{ type: 'text', text: 'admitted prompt' }])
  178. expect(staged[2]?.type === 'user/message' && staged[2].data.content)
  179. .toEqual([{ type: 'text', text: 'attached context' }])
  180. expect(staged[3]?.type === 'steering/message' && staged[3].data.content)
  181. .toEqual([{ type: 'text', text: 'admission steering' }])
  182. const request = JSON.stringify(adapter.requests[0]?.messages)
  183. expect(request).toContain('admitted prompt')
  184. expect(request).toContain('attached context')
  185. expect(request).toContain('admission steering')
  186. })
  187. it('keeps admission-time outbox input staged when admission is blocked', async () => {
  188. const adapter = new MockAdapter([textResponse('retried')])
  189. const ctx = await harness(adapter)
  190. const agent = ctx.agentLoop.create(SessionId('blocked-admission-outbox'), { provider: 'mock', model: 'mock' })
  191. const entered = Promise.withResolvers<undefined>()
  192. const decision = Promise.withResolvers<PromptDecision>()
  193. ctx.on('agent/prompt-submit', async () => {
  194. entered.resolve(undefined)
  195. return decision.promise
  196. })
  197. const blockedIdle = waitForIdle(ctx, agent)
  198. send(agent, 'blocked prompt')
  199. await entered.promise
  200. expect(agent.acceptsNextStep).toBe(true)
  201. agent.inject({
  202. content: [{ type: 'text', text: 'staged context' }],
  203. source: { kind: 'plugin', plugin: 'test' },
  204. })
  205. agent.steer({ content: [{ type: 'text', text: 'staged steering' }], source: { kind: 'user' } })
  206. decision.resolve({ kind: 'block', reason: 'policy' })
  207. await blockedIdle
  208. expect(agent.acceptsNextStep).toBe(false)
  209. expect(events(agent)).toEqual([])
  210. expect(adapter.requests).toEqual([])
  211. const retryIdle = waitForIdle(ctx, agent)
  212. agent.retry()
  213. await retryIdle
  214. const staged = events(agent).filter(event =>
  215. event.type === 'user/message' || event.type === 'steering/message')
  216. expect(staged.map(event => event.type)).toEqual(['user/message', 'steering/message'])
  217. expect(JSON.stringify(adapter.requests[0]?.messages)).not.toContain('blocked prompt')
  218. expect(JSON.stringify(adapter.requests[0]?.messages)).toContain('staged context')
  219. expect(JSON.stringify(adapter.requests[0]?.messages)).toContain('staged steering')
  220. })
  221. it('commits context-only injection when admission closes without a turn', async () => {
  222. const adapter = new MockAdapter([])
  223. const ctx = await harness(adapter)
  224. const agent = ctx.agentLoop.create(SessionId('blocked-admission-context'), { provider: 'mock', model: 'mock' })
  225. const entered = Promise.withResolvers<undefined>()
  226. const decision = Promise.withResolvers<PromptDecision>()
  227. ctx.on('agent/prompt-submit', async () => {
  228. entered.resolve(undefined)
  229. return decision.promise
  230. })
  231. const idle = waitForIdle(ctx, agent)
  232. send(agent, 'blocked prompt')
  233. await entered.promise
  234. agent.inject({
  235. content: [{ type: 'text', text: 'independent context' }],
  236. source: { kind: 'plugin', plugin: 'test' },
  237. })
  238. decision.resolve({ kind: 'block', reason: 'policy' })
  239. await idle
  240. const log = events(agent)
  241. expect(log.map(event => event.type)).toEqual(['user/message'])
  242. expect(log[0]?.type === 'user/message' && log[0].data.content)
  243. .toEqual([{ type: 'text', text: 'independent context' }])
  244. expect(adapter.requests).toEqual([])
  245. })
  246. it('retains rejected-admission context when its idle append fails', async () => {
  247. const adapter = new MockAdapter([textResponse('retried')])
  248. const ctx = await harness(adapter)
  249. const agent = ctx.agentLoop.create(SessionId('blocked-admission-append-failure'), {
  250. provider: 'mock',
  251. model: 'mock',
  252. })
  253. const warned = vi.spyOn(ctx.logger, 'warn').mockImplementation(() => undefined)
  254. vi.spyOn(agent.session, 'append').mockImplementationOnce(() => {
  255. throw new Error('append unavailable')
  256. })
  257. ctx.on('agent/prompt-submit', async () => ({ kind: 'block', reason: 'policy' }))
  258. agent.followup({ content: [{ type: 'text', text: 'blocked prompt' }], source: { kind: 'user' } })
  259. agent.inject({
  260. content: [{ type: 'text', text: 'retained context' }],
  261. source: { kind: 'plugin', plugin: 'test' },
  262. })
  263. await agent.whenIdle()
  264. expect(events(agent)).toEqual([])
  265. expect(warned).toHaveBeenCalledWith(expect.stringContaining('append unavailable'))
  266. const idle = waitForIdle(ctx, agent)
  267. agent.retry()
  268. await idle
  269. expect(events(agent).some(event => event.type === 'user/message'
  270. && JSON.stringify(event.data.content).includes('retained context'))).toBe(true)
  271. })
  272. it('adjacent blocked and allowed prompts keep independent turn outcomes', async () => {
  273. const adapter = new MockAdapter([textResponse('ran once')])
  274. const ctx = await harness(adapter)
  275. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  276. ctx.on('agent/prompt-submit', async (_agent, content, _source, _signal, next): Promise<PromptDecision> => {
  277. const text = content.map(b => (b.type === 'text' ? b.text : '')).join('')
  278. return text === 'secret' ? { kind: 'block', reason: 'policy: no secrets' } : next()
  279. })
  280. const reasons: TurnEndReason[] = []
  281. ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  282. // The rejected admission is dropped; the allowed prompt owns the only turn.
  283. send(agent, 'secret')
  284. send(agent, 'safe')
  285. await waitForIdle(ctx, agent)
  286. const log = events(agent)
  287. // The allowed prompt became a user/message and drove exactly one model call.
  288. const userMsgs = log.filter(e => e.type === 'user/message')
  289. expect(userMsgs).toHaveLength(1)
  290. expect(userMsgs[0]?.type === 'user/message' && userMsgs[0].data.content).toEqual([{ type: 'text', text: 'safe' }])
  291. expect(adapter.requests.length).toBeGreaterThanOrEqual(1)
  292. expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1)
  293. expect(reasons).toEqual([{ kind: 'completed' }])
  294. })
  295. it('a throwing prompt-submit listener drops that admission while an adjacent message survives', async () => {
  296. const adapter = new MockAdapter([textResponse('after')])
  297. const ctx = await harness(adapter)
  298. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  299. let threw = false
  300. ctx.on('agent/prompt-submit', async () => {
  301. if (!threw) { threw = true; throw new Error('prompt hook broke') }
  302. return { kind: 'allow' as const }
  303. })
  304. const errors: Error[] = []
  305. const reasons: TurnEndReason[] = []
  306. const statuses: string[] = []
  307. ctx.on('agent/error', (_a, _t, _s, error) => {
  308. if (error instanceof Error) errors.push(error)
  309. })
  310. ctx.on('agent/status', (subject, status) => { if (subject === agent) statuses.push(status) })
  311. ctx.on('session/event', (session, event) => {
  312. if (session === agent.session && event.type === 'turn/end') reasons.push(event.data.reason)
  313. })
  314. const idle = waitForIdle(ctx, agent)
  315. send(agent, 'first')
  316. send(agent, 'second')
  317. await idle
  318. expect(errors).toEqual([])
  319. const log = events(agent)
  320. expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1)
  321. expect(log.filter(e => e.type === 'turn/end')).toHaveLength(1)
  322. expect(reasons).toEqual([{ kind: 'completed' }])
  323. expect(statuses).toEqual(['running', 'idle'])
  324. expect(adapter.requests).toHaveLength(1)
  325. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('second')
  326. })
  327. })
  328. describe('agent/session-start', () => {
  329. it('fires once with source "startup" for a fresh create, before the first turn', async () => {
  330. const adapter = new MockAdapter([textResponse('ok')])
  331. const ctx = await harness(adapter)
  332. const sources: SessionStartSource[] = []
  333. ctx.on('agent/session-start', (_agent, source) => void sources.push(source))
  334. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  335. // fires synchronously at create, before any turn
  336. expect(sources).toEqual(['startup'])
  337. expect(events(agent).some(e => e.type === 'turn/start')).toBe(false)
  338. send(agent, 'go')
  339. await waitForIdle(ctx, agent)
  340. // still only one session-start
  341. expect(sources).toEqual(['startup'])
  342. })
  343. it('a session-start listener can inject context the first request sees', async () => {
  344. const adapter = new MockAdapter([textResponse('ok')])
  345. const ctx = await harness(adapter)
  346. ctx.on('agent/session-start', (agent) => {
  347. agent.inject({ content: [{ type: 'text', text: 'session preamble' }], source: { kind: 'plugin', plugin: 'test' } })
  348. })
  349. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  350. send(agent, 'go')
  351. await waitForIdle(ctx, agent)
  352. // the injected context reached the model on the first (only) request
  353. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('session preamble')
  354. // and is recorded with the plugin source, never mislabeled as a user prompt
  355. const ctxMsg = events(agent).find(e => e.type === 'user/message' && e.data.source.kind === 'plugin')
  356. expect(ctxMsg?.type === 'user/message' && ctxMsg.data.source).toEqual({ kind: 'plugin', plugin: 'test' })
  357. })
  358. it('a throwing session-start listener does not abort agent construction', async () => {
  359. const adapter = new MockAdapter([textResponse('ok')])
  360. const ctx = await harness(adapter)
  361. ctx.on('agent/session-start', () => { throw new Error('session-start hook broke') })
  362. // create must not throw — the listener error is contained/logged
  363. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  364. expect(agent.id).toBe(SessionId('a1'))
  365. // and the agent still runs
  366. send(agent, 'go')
  367. await waitForIdle(ctx, agent)
  368. expect(adapter.requests).toHaveLength(1)
  369. })
  370. })
  371. describe('tool additionalContexts buffering across a step', () => {
  372. it('appends each call\'s contexts only AFTER all tool/results, preserving adjacency', async () => {
  373. // One assistant step with TWO tool calls; the second model response stops.
  374. const twoCalls = [
  375. { type: 'block-start' as const, index: 0, blockType: 'tool-call' as const },
  376. { type: 'block-end' as const, index: 0, block: { type: 'tool-call' as const, id: CallId('c1'), name: 'echo', arguments: '{"text":"a"}' } },
  377. { type: 'block-start' as const, index: 1, blockType: 'tool-call' as const },
  378. { type: 'block-end' as const, index: 1, block: { type: 'tool-call' as const, id: CallId('c2'), name: 'echo', arguments: '{"text":"b"}' } },
  379. { type: 'usage' as const, usage: { inputTokens: 5, outputTokens: 5 } },
  380. { type: 'finish' as const, reason: { kind: 'tool-calls' as const } },
  381. ]
  382. const adapter = new MockAdapter([twoCalls, textResponse('done')])
  383. const ctx = await harness(adapter)
  384. ctx.tools.register(defineContentToolFixture({
  385. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  386. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  387. }))
  388. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  389. // Each call attaches one context naming itself.
  390. ctx.on('tools/post-execute', async (exec, _result): Promise<PostToolDecision> =>
  391. ({
  392. kind: 'accept',
  393. additionalContexts: [{
  394. content: [{ type: 'text', text: `ctx-${exec.callId}` }],
  395. source: { kind: 'plugin', plugin: 'p' },
  396. }],
  397. }))
  398. send(agent, 'go')
  399. await waitForIdle(ctx, agent)
  400. // Event order in the log: both tool/results, THEN both injected contexts —
  401. // never interleaved (which would break tool-call/result adjacency).
  402. const injected = events(agent).filter(e => e.type === 'user/message' && e.data.source.kind === 'plugin')
  403. const seqs = events(agent)
  404. const firstResult = seqs.findIndex(e => e.type === 'tool/result')
  405. const lastResult = seqs.map(e => e.type).lastIndexOf('tool/result')
  406. const firstCtx = seqs.findIndex(e => e === injected[0])
  407. expect(firstResult).toBeGreaterThanOrEqual(0)
  408. expect(lastResult).toBeGreaterThan(firstResult) // two results
  409. expect(firstCtx).toBeGreaterThan(lastResult) // context only after ALL results
  410. // both contexts present
  411. const ctxTexts = injected
  412. .flatMap(e => (e.type === 'user/message' ? e.data.content : []))
  413. .map(b => (b.type === 'text' ? b.text : ''))
  414. expect(ctxTexts).toEqual(['ctx-c1', 'ctx-c2'])
  415. })
  416. it('appends multiple contexts deferred by one composite tool after its outer result', async () => {
  417. const adapter = new MockAdapter([toolCallResponse('c1', 'composite', {}), textResponse('done')])
  418. const ctx = await harness(adapter)
  419. ctx.tools.register(defineContentToolFixture({
  420. name: 'composite', description: 'composite', parameters: {},
  421. async execute(_args, exec) {
  422. exec.deferContext({ content: [{ type: 'text', text: 'nested-a' }], source: { kind: 'plugin', plugin: 'a' } })
  423. exec.deferContext({ content: [{ type: 'text', text: 'nested-b' }], source: { kind: 'plugin', plugin: 'b' } })
  424. return [{ type: 'text', text: 'outer result' }]
  425. },
  426. }))
  427. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  428. send(agent, 'go')
  429. await waitForIdle(ctx, agent)
  430. const log = events(agent)
  431. const resultIndex = log.findIndex(event => event.type === 'tool/result')
  432. const contextEvents = log.filter(event => event.type === 'user/message' && event.data.source.kind === 'plugin')
  433. expect(resultIndex).toBeGreaterThanOrEqual(0)
  434. expect(log.findIndex(event => event === contextEvents[0])).toBeGreaterThan(resultIndex)
  435. expect(contextEvents.map(event => event.type === 'user/message' && event.data.source)).toEqual([
  436. { kind: 'plugin', plugin: 'a' },
  437. { kind: 'plugin', plugin: 'b' },
  438. ])
  439. })
  440. })
  441. describe('tools/pre-execute gate (native-plugin permission pattern, end-to-end through the loop)', () => {
  442. it('deny short-circuits dispatch into an isError result the model sees', async () => {
  443. const adapter = new MockAdapter([toolCallResponse('c1', 'danger', {}), textResponse('ok')])
  444. const ctx = await harness(adapter)
  445. let ran = false
  446. ctx.tools.register(defineContentToolFixture({
  447. name: 'danger', description: 'danger', parameters: {},
  448. async execute() { ran = true; return [{ type: 'text', text: 'should not run' }] },
  449. }))
  450. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  451. ctx.on('tools/pre-execute', async (exec, next): Promise<PreToolDecision> => {
  452. if (exec.name === 'danger') return { kind: 'deny', reason: 'blocked dangerous tool' }
  453. return next()
  454. })
  455. send(agent, 'go')
  456. await waitForIdle(ctx, agent)
  457. expect(ran).toBe(false)
  458. const result = events(agent).find(e => e.type === 'tool/result')
  459. expect(result?.type === 'tool/result' && result.data.isError).toBe(true)
  460. expect(result?.type === 'tool/result'
  461. && result.data.content.some(b => b.type === 'text' && b.text.includes('blocked dangerous tool'))).toBe(true)
  462. })
  463. })
  464. describe('worked example: a native hook plugin is just a cordis plugin on the seams', () => {
  465. // The whole point of the interception taxonomy: a "native hook" needs no dsh-hook-protocol,
  466. // no external command, no hook/* log — it is an ordinary cordis plugin subscribing to the
  467. // canonical events and returning typed decisions.
  468. const NativeGuard = {
  469. name: 'native-guard',
  470. apply(ctx: Context) {
  471. // 1. SessionStart: seed a standing instruction.
  472. ctx.on('agent/session-start', (agent, source) => {
  473. agent.inject({ content: [{ type: 'text', text: `policy active (started: ${source})` }], source: { kind: 'plugin', plugin: 'native-guard' } })
  474. })
  475. // 2. PromptSubmit: block a forbidden prompt, annotate the rest.
  476. ctx.on('agent/prompt-submit', async (_agent, content, _source, _signal, next): Promise<PromptDecision> => {
  477. const text = content.map(b => (b.type === 'text' ? b.text : '')).join('')
  478. if (text.includes('rm -rf')) return { kind: 'block', reason: 'destructive prompt blocked' }
  479. return next()
  480. })
  481. // 3. PreToolUse: deny a dangerous tool by name.
  482. ctx.on('tools/pre-execute', async (exec, next): Promise<PreToolDecision> => {
  483. if (exec.name === 'danger') return { kind: 'deny', reason: 'danger tool denied' }
  484. return next()
  485. })
  486. // 4. PostToolUse: attach context after a tool runs.
  487. ctx.on('tools/post-execute', async (_exec, _result, next): Promise<PostToolDecision> => {
  488. const decision = await next()
  489. if (decision.kind === 'accept') {
  490. return { kind: 'accept', additionalContexts: [{ content: [{ type: 'text', text: 'audited' }], source: { kind: 'plugin', plugin: 'native-guard' } }] }
  491. }
  492. return decision
  493. })
  494. },
  495. }
  496. it('all four seams fire for a real allowed turn with a tool call', async () => {
  497. const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'hi' }), textResponse('done')])
  498. const ctx = await harness(adapter)
  499. await ctx.plugin(NativeGuard)
  500. ctx.tools.register(defineContentToolFixture({
  501. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  502. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  503. }))
  504. const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
  505. send(agent, 'please echo hi')
  506. await waitForIdle(ctx, agent)
  507. const log = events(agent)
  508. // session-start preamble injected
  509. expect(log.some(e => e.type === 'user/message' && e.data.source.kind === 'plugin'
  510. && e.data.content.some(b => b.type === 'text' && b.text.includes('policy active (started: startup)')))).toBe(true)
  511. // prompt allowed → user-sourced user/message recorded
  512. expect(log.some(e => e.type === 'user/message' && e.data.source.kind === 'user')).toBe(true)
  513. // tool ran (echo allowed) and post-execute attached "audited" context
  514. expect(log.some(e => e.type === 'tool/result' && !e.data.isError)).toBe(true)
  515. expect(log.some(e => e.type === 'user/message' && e.data.source.kind === 'plugin'
  516. && e.data.content.some(b => b.type === 'text' && b.text === 'audited'))).toBe(true)
  517. // NO hook/* events — a native plugin needs none
  518. expect(log.some(e => e.type.startsWith('hook/'))).toBe(false)
  519. })
  520. it('the same plugin blocks a destructive prompt before a turn or model call', async () => {
  521. const adapter = new MockAdapter([textResponse('should not run')])
  522. const ctx = await harness(adapter)
  523. await ctx.plugin(NativeGuard)
  524. const agent = ctx.agentLoop.create(SessionId('a2'), { provider: 'mock', model: 'mock' })
  525. const reasons: TurnEndReason[] = []
  526. ctx.on('session/event', (_s, event: SessionEvent) => { if (event.type === 'turn/end') reasons.push(event.data.reason) })
  527. send(agent, 'run rm -rf /')
  528. await agent.whenIdle()
  529. expect(adapter.requests).toHaveLength(0)
  530. expect(reasons).toEqual([])
  531. })
  532. it('HMR-safety: disposing the plugin fiber removes all four listeners', async () => {
  533. const adapter = new MockAdapter([textResponse('ok')])
  534. const ctx = await harness(adapter)
  535. const fiber = await ctx.plugin(NativeGuard)
  536. await fiber.dispose()
  537. // After disposal, a destructive prompt is NOT blocked (the listener is gone).
  538. const agent = ctx.agentLoop.create(SessionId('a3'), { provider: 'mock', model: 'mock' })
  539. send(agent, 'run rm -rf /')
  540. await waitForIdle(ctx, agent)
  541. // the prompt ran (not rejected) — proving the prompt-submit listener was disposed
  542. expect(adapter.requests).toHaveLength(1)
  543. expect(events(agent).some(e => e.type === 'user/message')).toBe(true)
  544. })
  545. })