interception.spec.ts 20 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446
  1. import { describe, expect, it } from 'vitest'
  2. import { Context } from 'cordis'
  3. import LlmService, { CallId } from '@deepseek-ai/dsh-llm'
  4. import SessionStore, { type SessionEvent, type TurnEndReason } from '@deepseek-ai/dsh-session'
  5. import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
  6. import ToolRegistry, { defineTool, type PostToolDecision, type PreToolDecision } from '@deepseek-ai/dsh-tools'
  7. import AgentRegistry, {
  8. AgentId,
  9. type ContinuationDecision,
  10. type PromptDecision,
  11. type SessionStartSource,
  12. } from '@deepseek-ai/dsh-agent'
  13. import AgentLoop, { type ReactLoopAgent } from '@deepseek-ai/dsh-agent-loop'
  14. import { MockAdapter, textResponse, toolCallResponse } from './mock-adapter.ts'
  15. /**
  16. * The interception seams introduced by the hooks taxonomy: `agent/prompt-submit`,
  17. * `agent/session-start`, the reshaped `agent/turn-continuation`
  18. * ({@link ContinuationDecision}), and the `tools/pre-execute` / `tools/post-execute`
  19. * split with `additionalContext` buffering. These verify the canonical event
  20. * surface a hook bridge (or a native plugin) programs against, WITHOUT any
  21. * external protocol — a native plugin uses the typed decisions directly.
  22. */
  23. async function harness(adapter: MockAdapter) {
  24. const ctx = new Context()
  25. await ctx.plugin(LlmService)
  26. await ctx.plugin(SessionStore)
  27. await ctx.plugin(SystemPrompt)
  28. await ctx.plugin(ToolRegistry)
  29. await ctx.plugin(AgentRegistry)
  30. await ctx.plugin(AgentLoop, { agents: [] })
  31. ctx.llm.registerAdapter(['mock'], adapter)
  32. return ctx
  33. }
  34. function waitForIdle(ctx: Context, agent: ReactLoopAgent): Promise<void> {
  35. return new Promise((resolve) => {
  36. const dispose = ctx.on('agent/status', (subject, status) => {
  37. if (subject === agent && status === 'idle') {
  38. dispose()
  39. resolve()
  40. }
  41. })
  42. })
  43. }
  44. function send(agent: ReactLoopAgent, text: string) {
  45. agent.send([{ type: 'text', text }])
  46. }
  47. function events(agent: ReactLoopAgent): SessionEvent[] {
  48. return [...agent.session.events]
  49. }
  50. describe('agent/prompt-submit', () => {
  51. it('allow (default via next) records the user/message unchanged', async () => {
  52. const adapter = new MockAdapter([textResponse('ok')])
  53. const ctx = await harness(adapter)
  54. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  55. const seen: string[] = []
  56. ctx.on('agent/prompt-submit', async (_agent, content, _source, next) => {
  57. seen.push(content.map(b => (b.type === 'text' ? b.text : '')).join(''))
  58. return next()
  59. })
  60. send(agent, 'hello')
  61. await waitForIdle(ctx, agent)
  62. expect(seen).toEqual(['hello'])
  63. const userMsg = events(agent).find(e => e.type === 'user/message')
  64. expect(userMsg?.type === 'user/message' && userMsg.data.content).toEqual([{ type: 'text', text: 'hello' }])
  65. })
  66. it('allow with content REWRITES the prompt before it is recorded', async () => {
  67. const adapter = new MockAdapter([textResponse('ok')])
  68. const ctx = await harness(adapter)
  69. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  70. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  71. ({ kind: 'allow', content: [{ type: 'text', text: 'REWRITTEN' }] }))
  72. send(agent, 'original')
  73. await waitForIdle(ctx, agent)
  74. const userMsg = events(agent).find(e => e.type === 'user/message')
  75. expect(userMsg?.type === 'user/message' && userMsg.data.content).toEqual([{ type: 'text', text: 'REWRITTEN' }])
  76. // the rewritten prompt is what reached the model
  77. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('REWRITTEN')
  78. expect(JSON.stringify(adapter.requests[0]!.messages)).not.toContain('original')
  79. })
  80. it('allow with additionalContext injects a separate context/message into the turn', async () => {
  81. const adapter = new MockAdapter([textResponse('ok')])
  82. const ctx = await harness(adapter)
  83. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  84. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  85. ({
  86. kind: 'allow',
  87. additionalContext: { content: [{ type: 'text', text: 'extra ctx' }], source: { kind: 'plugin', plugin: 'test' } },
  88. }))
  89. send(agent, 'go')
  90. await waitForIdle(ctx, agent)
  91. const log = events(agent)
  92. const userMsg = log.find(e => e.type === 'user/message')
  93. const ctxMsg = log.find(e => e.type === 'context/message')
  94. expect(userMsg).toBeDefined()
  95. expect(ctxMsg?.type === 'context/message' && ctxMsg.data.content).toEqual([{ type: 'text', text: 'extra ctx' }])
  96. expect(ctxMsg?.type === 'context/message' && ctxMsg.data.source).toEqual({ kind: 'plugin', plugin: 'test' })
  97. // both the prompt and the injected context reach the model
  98. const sent = JSON.stringify(adapter.requests[0]!.messages)
  99. expect(sent).toContain('extra ctx')
  100. })
  101. it('block drops the (only) prompt → zero-step turn ends rejected, model never called', async () => {
  102. const adapter = new MockAdapter([textResponse('should not run')])
  103. const ctx = await harness(adapter)
  104. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  105. ctx.on('agent/prompt-submit', async (): Promise<PromptDecision> =>
  106. ({ kind: 'block', reason: 'blocked by policy' }))
  107. const reasons: TurnEndReason[] = []
  108. ctx.on('agent/turn-end', (_a, _t, reason) => void reasons.push(reason))
  109. send(agent, 'do something')
  110. await waitForIdle(ctx, agent)
  111. // the model was never called
  112. expect(adapter.requests).toHaveLength(0)
  113. // the turn opened and closed balanced, with no user/message and no step
  114. const log = events(agent)
  115. expect(log.some(e => e.type === 'turn/start')).toBe(true)
  116. expect(log.some(e => e.type === 'turn/end')).toBe(true)
  117. expect(log.some(e => e.type === 'user/message')).toBe(false)
  118. expect(log.some(e => e.type === 'step/start')).toBe(false)
  119. // ended rejected with the block reason
  120. expect(reasons).toEqual([{ kind: 'rejected', reason: 'blocked by policy' }])
  121. const turnEnd = log.findLast(e => e.type === 'turn/end')
  122. expect(turnEnd?.type === 'turn/end' && turnEnd.data.reason).toEqual({ kind: 'rejected', reason: 'blocked by policy' })
  123. })
  124. it('a throwing prompt-submit listener ends the turn balanced (error), loop survives', async () => {
  125. const adapter = new MockAdapter([textResponse('after')])
  126. const ctx = await harness(adapter)
  127. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  128. let threw = false
  129. ctx.on('agent/prompt-submit', async () => {
  130. if (!threw) { threw = true; throw new Error('prompt hook broke') }
  131. return { kind: 'allow' as const }
  132. })
  133. const errors: Error[] = []
  134. ctx.on('agent/error', (_a, _t, _s, error) => void errors.push(error))
  135. send(agent, 'first')
  136. await waitForIdle(ctx, agent)
  137. expect(errors.map(e => e.message)).toEqual(['prompt hook broke'])
  138. // turn balanced
  139. const log = events(agent)
  140. expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1)
  141. expect(log.filter(e => e.type === 'turn/end')).toHaveLength(1)
  142. // loop survives: a second prompt runs normally
  143. send(agent, 'second')
  144. await waitForIdle(ctx, agent)
  145. expect(adapter.requests.length).toBeGreaterThanOrEqual(1)
  146. })
  147. })
  148. describe('agent/session-start', () => {
  149. it('fires once with source "startup" for a fresh create, before the first turn', async () => {
  150. const adapter = new MockAdapter([textResponse('ok')])
  151. const ctx = await harness(adapter)
  152. const sources: SessionStartSource[] = []
  153. ctx.on('agent/session-start', (_agent, source) => void sources.push(source))
  154. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  155. // fires synchronously at create, before any turn
  156. expect(sources).toEqual(['startup'])
  157. expect(events(agent).some(e => e.type === 'turn/start')).toBe(false)
  158. send(agent, 'go')
  159. await waitForIdle(ctx, agent)
  160. // still only one session-start
  161. expect(sources).toEqual(['startup'])
  162. })
  163. it('a session-start listener can inject context the first request sees', async () => {
  164. const adapter = new MockAdapter([textResponse('ok')])
  165. const ctx = await harness(adapter)
  166. ctx.on('agent/session-start', (agent) => {
  167. agent.inject([{ type: 'text', text: 'session preamble' }], { source: { kind: 'plugin', plugin: 'test' } })
  168. })
  169. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  170. send(agent, 'go')
  171. await waitForIdle(ctx, agent)
  172. // the injected context reached the model on the first (only) request
  173. expect(JSON.stringify(adapter.requests[0]!.messages)).toContain('session preamble')
  174. // and is recorded with the plugin source, never mislabeled as a user prompt
  175. const ctxMsg = events(agent).find(e => e.type === 'context/message')
  176. expect(ctxMsg?.type === 'context/message' && ctxMsg.data.source).toEqual({ kind: 'plugin', plugin: 'test' })
  177. })
  178. it('a throwing session-start listener does not abort agent construction', async () => {
  179. const adapter = new MockAdapter([textResponse('ok')])
  180. const ctx = await harness(adapter)
  181. ctx.on('agent/session-start', () => { throw new Error('session-start hook broke') })
  182. // create must not throw — the listener error is contained/logged
  183. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  184. expect(agent.id).toBe(AgentId('a1'))
  185. // and the agent still runs
  186. send(agent, 'go')
  187. await waitForIdle(ctx, agent)
  188. expect(adapter.requests).toHaveLength(1)
  189. })
  190. })
  191. describe('agent/turn-continuation (ContinuationDecision)', () => {
  192. it('a continue decision with a reason records next-step steering in the same turn', async () => {
  193. const adapter = new MockAdapter([textResponse('step 1 no tools'), textResponse('step 2')])
  194. const ctx = await harness(adapter)
  195. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  196. let forced = false
  197. ctx.on('agent/turn-continuation', async (_agent, _turn, _default, next): Promise<ContinuationDecision> => {
  198. if (!forced) {
  199. forced = true
  200. return { action: 'continue', reason: { content: [{ type: 'text', text: 'keep going on the goal' }], source: { kind: 'plugin', plugin: 'goal' } } }
  201. }
  202. return next()
  203. })
  204. send(agent, 'go')
  205. await waitForIdle(ctx, agent)
  206. const log = events(agent)
  207. // same turn, two steps
  208. expect(log.filter(e => e.type === 'turn/start')).toHaveLength(1)
  209. expect(log.filter(e => e.type === 'step/start')).toHaveLength(2)
  210. // the reason was recorded as steering BEFORE step 2, with its plugin source
  211. const steering = log.find(e => e.type === 'steering/message')
  212. expect(steering?.type === 'steering/message' && steering.data.content).toEqual([{ type: 'text', text: 'keep going on the goal' }])
  213. expect(steering?.type === 'steering/message' && steering.data.source).toEqual({ kind: 'plugin', plugin: 'goal' })
  214. // and reached the next request
  215. expect(JSON.stringify(adapter.requests[1]!.messages)).toContain('keep going on the goal')
  216. })
  217. it('a stop decision ends the turn even when the step had tool calls', async () => {
  218. const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'hi' })])
  219. const ctx = await harness(adapter)
  220. ctx.tools.register(defineTool({
  221. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  222. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  223. }))
  224. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  225. ctx.on('agent/turn-continuation', async (): Promise<ContinuationDecision> => ({ action: 'stop' }))
  226. send(agent, 'go')
  227. await waitForIdle(ctx, agent)
  228. // default would have continued (had tool calls), but the stop decision wins
  229. expect(adapter.requests).toHaveLength(1)
  230. expect(events(agent).some(e => e.type === 'tool/result')).toBe(true)
  231. })
  232. })
  233. describe('tools/post-execute additionalContext buffering across a multi-call step', () => {
  234. it('appends each call\'s additionalContext only AFTER all tool/results, preserving adjacency', async () => {
  235. // One assistant step with TWO tool calls; the second model response stops.
  236. const twoCalls = [
  237. { type: 'block-start' as const, index: 0, blockType: 'tool-call' as const },
  238. { type: 'block-end' as const, index: 0, block: { type: 'tool-call' as const, id: CallId('c1'), name: 'echo', arguments: '{"text":"a"}' } },
  239. { type: 'block-start' as const, index: 1, blockType: 'tool-call' as const },
  240. { type: 'block-end' as const, index: 1, block: { type: 'tool-call' as const, id: CallId('c2'), name: 'echo', arguments: '{"text":"b"}' } },
  241. { type: 'usage' as const, usage: { inputTokens: 5, outputTokens: 5 } },
  242. { type: 'finish' as const, reason: { kind: 'tool-calls' as const } },
  243. ]
  244. const adapter = new MockAdapter([twoCalls, textResponse('done')])
  245. const ctx = await harness(adapter)
  246. ctx.tools.register(defineTool({
  247. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  248. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  249. }))
  250. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  251. // Each call attaches additionalContext naming itself.
  252. ctx.on('tools/post-execute', async (exec, _result): Promise<PostToolDecision> =>
  253. ({ kind: 'accept', additionalContext: { content: [{ type: 'text', text: `ctx-${exec.callId}` }], source: { kind: 'plugin', plugin: 'p' } } }))
  254. send(agent, 'go')
  255. await waitForIdle(ctx, agent)
  256. // Event order in the log: both tool/results, THEN both context/messages —
  257. // never interleaved (which would break tool-call/result adjacency).
  258. const types = events(agent).map(e => e.type)
  259. const firstResult = types.indexOf('tool/result')
  260. const lastResult = types.lastIndexOf('tool/result')
  261. const firstCtx = types.indexOf('context/message')
  262. expect(firstResult).toBeGreaterThanOrEqual(0)
  263. expect(lastResult).toBeGreaterThan(firstResult) // two results
  264. expect(firstCtx).toBeGreaterThan(lastResult) // context only after ALL results
  265. // both contexts present
  266. const ctxTexts = events(agent)
  267. .filter(e => e.type === 'context/message')
  268. .flatMap(e => (e.type === 'context/message' ? e.data.content : []))
  269. .map(b => (b.type === 'text' ? b.text : ''))
  270. expect(ctxTexts).toEqual(['ctx-c1', 'ctx-c2'])
  271. })
  272. })
  273. describe('tools/pre-execute gate (native-plugin permission pattern, end-to-end through the loop)', () => {
  274. it('deny short-circuits dispatch into an isError result the model sees', async () => {
  275. const adapter = new MockAdapter([toolCallResponse('c1', 'danger', {}), textResponse('ok')])
  276. const ctx = await harness(adapter)
  277. let ran = false
  278. ctx.tools.register(defineTool({
  279. name: 'danger', description: 'danger', parameters: {},
  280. async execute() { ran = true; return [{ type: 'text', text: 'should not run' }] },
  281. }))
  282. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  283. ctx.on('tools/pre-execute', async (exec, next): Promise<PreToolDecision> => {
  284. if (exec.name === 'danger') return { kind: 'deny', reason: 'blocked dangerous tool' }
  285. return next()
  286. })
  287. send(agent, 'go')
  288. await waitForIdle(ctx, agent)
  289. expect(ran).toBe(false)
  290. const result = events(agent).find(e => e.type === 'tool/result')
  291. expect(result?.type === 'tool/result' && result.data.isError).toBe(true)
  292. expect(result?.type === 'tool/result'
  293. && result.data.content.some(b => b.type === 'text' && b.text.includes('blocked dangerous tool'))).toBe(true)
  294. })
  295. })
  296. describe('worked example: a native hook plugin is just a cordis plugin on the seams', () => {
  297. // The whole point of the interception taxonomy: a "native hook" needs no
  298. // dsh-hook-protocol, no external command, no hook/* log — it is an ordinary
  299. // cordis plugin subscribing to the canonical events and returning typed
  300. // decisions. This proves all four seams compose end-to-end through the REAL
  301. // loop, with NO hook/* SessionEvents involved (those belong to the bridge lib).
  302. const NativeGuard = {
  303. name: 'native-guard',
  304. apply(ctx: Context) {
  305. // 1. SessionStart: seed a standing instruction.
  306. ctx.on('agent/session-start', (agent, source) => {
  307. agent.inject(
  308. [{ type: 'text', text: `policy active (started: ${source})` }],
  309. { source: { kind: 'plugin', plugin: 'native-guard' } },
  310. )
  311. })
  312. // 2. PromptSubmit: block a forbidden prompt, annotate the rest.
  313. ctx.on('agent/prompt-submit', async (_agent, content, _source, next): Promise<PromptDecision> => {
  314. const text = content.map(b => (b.type === 'text' ? b.text : '')).join('')
  315. if (text.includes('rm -rf')) return { kind: 'block', reason: 'destructive prompt blocked' }
  316. return next()
  317. })
  318. // 3. PreToolUse: deny a dangerous tool by name.
  319. ctx.on('tools/pre-execute', async (exec, next): Promise<PreToolDecision> => {
  320. if (exec.name === 'danger') return { kind: 'deny', reason: 'danger tool denied' }
  321. return next()
  322. })
  323. // 4. PostToolUse: attach context after a tool runs.
  324. ctx.on('tools/post-execute', async (_exec, _result, next): Promise<PostToolDecision> => {
  325. const decision = await next()
  326. if (decision.kind === 'accept') {
  327. return { kind: 'accept', additionalContext: { content: [{ type: 'text', text: 'audited' }], source: { kind: 'plugin', plugin: 'native-guard' } } }
  328. }
  329. return decision
  330. })
  331. },
  332. }
  333. it('all four seams fire for a real allowed turn with a tool call', async () => {
  334. const adapter = new MockAdapter([toolCallResponse('c1', 'echo', { text: 'hi' }), textResponse('done')])
  335. const ctx = await harness(adapter)
  336. await ctx.plugin(NativeGuard)
  337. ctx.tools.register(defineTool({
  338. name: 'echo', description: 'echo', parameters: { text: { type: 'string' } },
  339. async execute(args) { return [{ type: 'text', text: String(args.text) }] },
  340. }))
  341. const agent = ctx.agentLoop.create(AgentId('a1'), { model: 'mock' })
  342. send(agent, 'please echo hi')
  343. await waitForIdle(ctx, agent)
  344. const log = events(agent)
  345. // session-start preamble injected
  346. expect(log.some(e => e.type === 'context/message'
  347. && e.data.content.some(b => b.type === 'text' && b.text.includes('policy active (started: startup)')))).toBe(true)
  348. // prompt allowed → user/message recorded
  349. expect(log.some(e => e.type === 'user/message')).toBe(true)
  350. // tool ran (echo allowed) and post-execute attached "audited" context
  351. expect(log.some(e => e.type === 'tool/result' && !e.data.isError)).toBe(true)
  352. expect(log.some(e => e.type === 'context/message'
  353. && e.data.content.some(b => b.type === 'text' && b.text === 'audited'))).toBe(true)
  354. // NO hook/* events — a native plugin needs none
  355. expect(log.some(e => e.type.startsWith('hook/'))).toBe(false)
  356. })
  357. it('the same plugin blocks a destructive prompt → rejected turn, model never called', async () => {
  358. const adapter = new MockAdapter([textResponse('should not run')])
  359. const ctx = await harness(adapter)
  360. await ctx.plugin(NativeGuard)
  361. const agent = ctx.agentLoop.create(AgentId('a2'), { model: 'mock' })
  362. const reasons: TurnEndReason[] = []
  363. ctx.on('agent/turn-end', (_a, _t, reason) => void reasons.push(reason))
  364. send(agent, 'run rm -rf /')
  365. await waitForIdle(ctx, agent)
  366. expect(adapter.requests).toHaveLength(0)
  367. expect(reasons).toEqual([{ kind: 'rejected', reason: 'destructive prompt blocked' }])
  368. })
  369. it('HMR-safety: disposing the plugin fiber removes all four listeners', async () => {
  370. const adapter = new MockAdapter([textResponse('ok')])
  371. const ctx = await harness(adapter)
  372. const fiber = await ctx.plugin(NativeGuard)
  373. await fiber.dispose()
  374. // After disposal, a destructive prompt is NOT blocked (the listener is gone).
  375. const agent = ctx.agentLoop.create(AgentId('a3'), { model: 'mock' })
  376. send(agent, 'run rm -rf /')
  377. await waitForIdle(ctx, agent)
  378. // the prompt ran (not rejected) — proving the prompt-submit listener was disposed
  379. expect(adapter.requests).toHaveLength(1)
  380. expect(events(agent).some(e => e.type === 'user/message')).toBe(true)
  381. })
  382. })