| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395 |
- import { describe, expect, it } from 'vitest'
- import { Context } from 'cordis'
- import { CallId } from '@deepseek-ai/dsh-llm'
- import { SessionId, type SessionEvent } from '@deepseek-ai/dsh-session'
- import { defineContentToolFixture } from '@deepseek-ai/dsh-tools'
- import type { Agent } from '@deepseek-ai/dsh-agent'
- import AgentLoop from '@deepseek-ai/dsh-agent-loop'
- import { mountAgentLoopTestDependencies } from '@deepseek-ai/dsh-agent-loop-testkit'
- import * as RepeatToolGuard from '@deepseek-ai/dsh-repeat-tool-guard'
- import type { Config } from '@deepseek-ai/dsh-repeat-tool-guard'
- import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
- const testToolSignal = new AbortController().signal
- /**
- * Behavior suite for the repeat-tool-call guard: chain semantics (identical /
- * different-tracked / untracked-transparent / per-agent / resets), threshold
- * escalation incl. the `thresholds[0]` gentle-text rule, canonicalization,
- * fold-onto-downstream-decision, and fail-loud config validation — all driven
- * through a real agent loop against a scripted mock adapter (no network).
- */
- /** Boot the core spine + the guard; the caller registers adapters and extra listeners. */
- async function harness(config: Config = {}): Promise<Context> {
- const ctx = new Context()
- await mountAgentLoopTestDependencies(ctx)
- await ctx.plugin(AgentLoop, { agents: [] })
- await ctx.plugin(RepeatToolGuard, config)
- ctx.tools.register(defineContentToolFixture({ name: 'probe', description: 'p', parameters: {}, async execute() { return [{ type: 'text', text: 'ok' }] } }))
- ctx.tools.register(defineContentToolFixture({ name: 'other', description: 'o', parameters: {}, async execute() { return [{ type: 'text', text: 'ok' }] } }))
- return ctx
- }
- function waitForIdle(ctx: Context, agent: Agent): Promise<void> {
- return new Promise((resolve) => { const d = ctx.on('agent/status', (s, st) => { if (s === agent && st === 'idle') { d(); resolve() } }) })
- }
- /** Every injected-context user message in the agent's log, flattened to joined text + source for terse assertions. */
- function reminders(agent: Agent): { text: string; source: unknown }[] {
- return [...agent.session.events]
- .filter((e): e is SessionEvent<'user/message'> => e.type === 'user/message' && e.data.source.kind !== 'user')
- .map(e => ({
- text: e.data.content.map(block => block.type === 'text' ? block.text : '').join('|'),
- source: e.data.source,
- }))
- }
- const GUARD_SOURCE = { kind: 'plugin', plugin: 'repeat-tool-guard' }
- describe('threshold escalation', () => {
- it('reminds gently at the first default threshold (3) and in detail at the second (5)', async () => {
- const ctx = await harness()
- const adapter = new MockAdapter([
- ...Array.from({ length: 5 }, (_, i) => toolCallResponse(`c${i}`, 'probe', { q: 'same' })),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(2)
- expect(found[0]!.text).toContain('repeating the exact same tool call')
- expect(found[0]!.source).toEqual(GUARD_SOURCE)
- expect(found[1]!.text).toContain('consecutive_calls: 5')
- expect(found[1]!.text).toContain('- tool: probe')
- expect(found[1]!.text).toContain('{"q":"same"}')
- expect(found[1]!.source).toEqual(GUARD_SOURCE)
- })
- it('keys the gentle text to thresholds[0], not the literal 3', async () => {
- const ctx = await harness({ thresholds: [4, 2] }) // unsorted on purpose: normalized ascending
- const adapter = new MockAdapter([
- ...Array.from({ length: 4 }, (_, i) => toolCallResponse(`c${i}`, 'probe', {})),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(2)
- expect(found[0]!.text).toContain('repeating the exact same tool call') // gentle at 2
- expect(found[1]!.text).toContain('consecutive_calls: 4') // detailed at 4
- })
- })
- describe('chain semantics', () => {
- it('caps the detailed reminder arguments at argumentsPreviewChars (detection still keys on the full string)', async () => {
- const ctx = await harness({ thresholds: [2, 3], argumentsPreviewChars: 24 })
- const bigPayload = 'x'.repeat(400)
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { body: bigPayload }),
- toolCallResponse('c2', 'probe', { body: bigPayload }),
- toolCallResponse('c3', 'probe', { body: bigPayload }),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(2) // gentle at 2, detailed at 3 — full-key matching survived the cap
- const detailed = found[1]!.text
- expect(detailed).toContain('- arguments: {"body":"xxxxxxxxxxxxxx') // 24-char head
- expect(detailed).toContain('… (+387 more chars)')
- expect(detailed).not.toContain(bigPayload)
- })
- it('a different tracked call resets the chain', async () => {
- const ctx = await harness()
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- toolCallResponse('c2', 'probe', { q: 1 }),
- toolCallResponse('c3', 'other', {}), // tracked, different → reset
- toolCallResponse('c4', 'probe', { q: 1 }),
- toolCallResponse('c5', 'probe', { q: 1 }),
- toolCallResponse('c6', 'probe', { q: 1 }), // 3rd consecutive AFTER the reset
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- expect(reminders(agent)).toHaveLength(1)
- })
- it('excluded calls are transparent: they neither count nor reset', async () => {
- const ctx = await harness({ exclude: ['other'] })
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- toolCallResponse('c2', 'other', {}), // excluded → invisible to the chain
- toolCallResponse('c3', 'probe', { q: 1 }),
- toolCallResponse('c4', 'other', {}),
- toolCallResponse('c5', 'probe', { q: 1 }), // 3rd consecutive probe
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(1)
- expect(found[0]!.text).toContain('repeating the exact same tool call')
- })
- it('include patterns track only matching tools (wildcard star)', async () => {
- const ctx = await harness({ include: ['pro*'] })
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'other', {}),
- toolCallResponse('c2', 'other', {}),
- toolCallResponse('c3', 'other', {}), // 3 identical, but untracked
- toolCallResponse('c4', 'probe', {}),
- toolCallResponse('c5', 'probe', {}),
- toolCallResponse('c6', 'probe', {}), // 3 identical, tracked
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(1)
- expect(found[0]!.text).toContain('repeating the exact same tool call')
- })
- it('escapes regex metacharacters in patterns (a dot matches only a literal dot)', async () => {
- const ctx = await harness({ exclude: ['pr.be'] }) // would match 'probe' as a regex; must not as a wildcard
- const adapter = new MockAdapter([
- ...Array.from({ length: 3 }, (_, i) => toolCallResponse(`c${i}`, 'probe', {})),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- expect(reminders(agent)).toHaveLength(1) // probe was NOT excluded
- })
- it('canonicalization ignores property order, deeply', async () => {
- const ctx = await harness()
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { a: 1, nested: { x: [1, 2], y: null } }),
- toolCallResponse('c2', 'probe', { nested: { y: null, x: [1, 2] }, a: 1 }),
- toolCallResponse('c3', 'probe', { a: 1, nested: { x: [1, 2], y: null } }),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- expect(reminders(agent)).toHaveLength(1) // all three canonicalize identically
- })
- it('keys chains per agent: one agent repeating never trips another', async () => {
- const ctx = await harness()
- ctx.llm.registerAdapter(['mock-a'], new MockAdapter([
- toolCallResponse('a1', 'probe', { q: 1 }),
- toolCallResponse('a2', 'probe', { q: 1 }),
- textResponse('done'),
- ]))
- ctx.llm.registerAdapter(['mock-b'], new MockAdapter([
- toolCallResponse('b1', 'probe', { q: 1 }),
- toolCallResponse('b2', 'probe', { q: 1 }),
- toolCallResponse('b3', 'probe', { q: 1 }),
- textResponse('done'),
- ]))
- const agentA = ctx.agentLoop.create(SessionId('a'), { provider: 'mock-a', model: 'model-a' })
- const agentB = ctx.agentLoop.create(SessionId('b'), { provider: 'mock-b', model: 'model-b' })
- agentA.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- agentB.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await Promise.all([waitForIdle(ctx, agentA), waitForIdle(ctx, agentB)])
- expect(reminders(agentA)).toHaveLength(0) // 2 repeats < 3, despite B's 3 in the same registry
- expect(reminders(agentB)).toHaveLength(1)
- })
- it('a new user prompt resets the chain', async () => {
- const ctx = await harness()
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- toolCallResponse('c2', 'probe', { q: 1 }),
- textResponse('turn one done'),
- toolCallResponse('c3', 'probe', { q: 1 }), // without the reset this would be the 3rd
- textResponse('turn two done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- agent.followup({ content: [{ type: 'text', text: 'again' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- expect(reminders(agent)).toHaveLength(0)
- })
- it('drops an agent chain on disposal', async () => {
- const ctx = await harness({ thresholds: [2] })
- ctx.llm.registerAdapter(['mock'], new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- textResponse('done'),
- toolCallResponse('c2', 'probe', { q: 1 }), // same id, fresh agent: count 1, not 2
- textResponse('done'),
- ]))
- // Loop agents are torn down by disposing the scope that created them
- // (the loop.spec pattern): a child plugin fiber owns `first`.
- let first!: Agent
- const fiber = await ctx.plugin(Object.assign((inner: Context) => {
- first = inner.agentLoop.create(SessionId('reused'), { provider: 'mock', model: 'mock' })
- }, { inject: ['agentLoop'] }))
- first.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, first)
- await fiber.dispose()
- await first.whenIdle()
- const second = ctx.agentLoop.create(SessionId('reused'), { provider: 'mock', model: 'mock' })
- second.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, second)
- expect(reminders(second)).toHaveLength(0)
- })
- it('counts denied calls: hammering a denied tool still draws the reminder', async () => {
- const ctx = await harness({ thresholds: [2] })
- ctx.on('tools/pre-execute', async () => ({ kind: 'deny' as const, reason: 'sealed' }))
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- toolCallResponse('c2', 'probe', { q: 1 }),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- expect(reminders(agent)).toHaveLength(1)
- })
- it('ignores direct executes with no agent (they neither crash nor advance any chain)', async () => {
- const ctx = await harness({ thresholds: [2] })
- const direct = await ctx.tools.execute({ signal: testToolSignal, callId: CallId('d1'), name: 'probe', arguments: { q: 1 } })
- expect(direct.isError).toBe(false)
- ctx.llm.registerAdapter(['mock'], new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }), // if the direct call had counted, this would be #2
- textResponse('done'),
- ]))
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- expect(reminders(agent)).toHaveLength(0)
- })
- })
- describe('fold onto the downstream decision', () => {
- it('folds the reminder onto a downstream block and keeps its feedback', async () => {
- const ctx = await harness({ thresholds: [2] })
- ctx.on('tools/post-execute', async () => ({
- kind: 'block' as const,
- feedback: [{ type: 'text' as const, text: 'nope' }],
- additionalContexts: [{ content: [{ type: 'text' as const, text: 'downstream-ctx' }], source: { kind: 'plugin' as const, plugin: 'test' } }],
- }))
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- toolCallResponse('c2', 'probe', { q: 1 }),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(3)
- // Call 1: below threshold — the downstream context passes through untouched.
- expect(found[0]!.text).toBe('downstream-ctx')
- expect(found[0]!.source).toEqual({ kind: 'plugin', plugin: 'test' })
- // Call 2: reminder and downstream context retain separate provenance.
- expect(found[1]!.text).toContain('repeating the exact same tool call')
- expect(found[1]!.source).toEqual(GUARD_SOURCE)
- expect(found[2]).toEqual({ text: 'downstream-ctx', source: { kind: 'plugin', plugin: 'test' } })
- // The block's feedback reached the tool result unchanged.
- const results = [...agent.session.events].filter((e): e is SessionEvent<'tool/result'> => e.type === 'tool/result')
- expect(results.every(r => r.data.isError)).toBe(true)
- expect(results[1]!.data.content).toEqual([{ type: 'text', text: 'nope' }])
- })
- it('preserves a downstream canonical value replacement while folding', async () => {
- const ctx = await harness({ thresholds: [2] })
- ctx.on('tools/post-execute', async () => ({
- kind: 'accept' as const,
- value: [{ type: 'text' as const, text: 'replaced' }],
- }))
- const adapter = new MockAdapter([
- toolCallResponse('c1', 'probe', { q: 1 }),
- toolCallResponse('c2', 'probe', { q: 1 }),
- textResponse('done'),
- ])
- ctx.llm.registerAdapter(['mock'], adapter)
- const agent = ctx.agentLoop.create(SessionId('a1'), { provider: 'mock', model: 'mock' })
- agent.followup({ content: [{ type: 'text', text: 'go' }], source: { kind: 'user' } })
- await waitForIdle(ctx, agent)
- const found = reminders(agent)
- expect(found).toHaveLength(1)
- expect(found[0]!.text).toContain('repeating the exact same tool call')
- const results = [...agent.session.events].filter((e): e is SessionEvent<'tool/result'> => e.type === 'tool/result')
- expect(results[1]!.data.content).toEqual([{ type: 'text', text: 'replaced' }])
- })
- })
- describe('config validation fails loud', () => {
- async function spine(): Promise<Context> {
- const ctx = new Context()
- await mountAgentLoopTestDependencies(ctx)
- await ctx.plugin(AgentLoop, { agents: [] })
- return ctx
- }
- it('rejects an empty thresholds list', async () => {
- const ctx = await spine()
- await expect(ctx.plugin(RepeatToolGuard, { thresholds: [] })).rejects.toThrow(/must not be empty/)
- })
- it('rejects a threshold below 2', async () => {
- const ctx = await spine()
- await expect(ctx.plugin(RepeatToolGuard, { thresholds: [1, 3] })).rejects.toThrow(/integer >= 2/)
- })
- it('rejects a non-integer threshold', async () => {
- const ctx = await spine()
- await expect(ctx.plugin(RepeatToolGuard, { thresholds: [2.5] })).rejects.toThrow(/integer >= 2/)
- })
- it('rejects duplicate thresholds', async () => {
- const ctx = await spine()
- await expect(ctx.plugin(RepeatToolGuard, { thresholds: [3, 3] })).rejects.toThrow(/duplicates/)
- })
- it('rejects a non-positive or fractional argumentsPreviewChars', async () => {
- const ctx = await spine()
- await expect(ctx.plugin(RepeatToolGuard, { argumentsPreviewChars: 0 })).rejects.toThrow(/argumentsPreviewChars/)
- const ctx2 = await spine()
- await expect(ctx2.plugin(RepeatToolGuard, { argumentsPreviewChars: 12.5 })).rejects.toThrow(/argumentsPreviewChars/)
- })
- })
|