|
|
@@ -1,5 +1,6 @@
|
|
|
-import { describe, expect, it } from 'vitest'
|
|
|
+import { describe, expect, it, vi } from 'vitest'
|
|
|
import { Context } from 'cordis'
|
|
|
+import TurndownService from 'turndown'
|
|
|
import { CallId } from '@deepseek-ai/dsh-llm'
|
|
|
import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
|
|
|
import ToolRegistry, { type ToolExecutionResult } from '@deepseek-ai/dsh-tools'
|
|
|
@@ -13,8 +14,6 @@ import {
|
|
|
parseFetchArgs,
|
|
|
presentSearchCall,
|
|
|
presentFetchCall,
|
|
|
- renderBody,
|
|
|
- htmlToMarkdown,
|
|
|
WEB_SEARCH_MAX_RESULTS,
|
|
|
} from '@deepseek-ai/dsh-tool-web'
|
|
|
|
|
|
@@ -82,17 +81,29 @@ describe('search formatting', () => {
|
|
|
expect(parseSearchArgs({ query: 'hi' })).toEqual({ query: 'hi' })
|
|
|
})
|
|
|
|
|
|
+ it('falls back to the raw URL as a source label when the URL is unparseable', () => {
|
|
|
+ const out = formatSearchOutput({ truncated: false, sources: [{ url: 'not a url' }] })
|
|
|
+ expect(out).toContain('[not a url](not a url)')
|
|
|
+ })
|
|
|
+
|
|
|
it('presents a search call as a search-kind card titled by the query', () => {
|
|
|
expect(presentSearchCall({ query: 'find me' })).toEqual({ card: 'generic', title: 'find me', kind: 'search', rawInput: 'find me' })
|
|
|
})
|
|
|
})
|
|
|
|
|
|
describe('fetch formatting', () => {
|
|
|
+ const NO_CAP = 1_000_000
|
|
|
+ const HEADER = 'Fetched https://a.test (HTTP 200)\n\n'
|
|
|
+ const renderHtml = (content: string) => formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content },
|
|
|
+ }, NO_CAP).slice(HEADER.length)
|
|
|
+
|
|
|
it('renders an html body to markdown text with a status header', () => {
|
|
|
const out = formatFetchOutput({
|
|
|
url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
body: { kind: 'html', content: '<h1>Title</h1><p>Body text</p>' },
|
|
|
- })
|
|
|
+ }, NO_CAP)
|
|
|
expect(out).toContain('Fetched https://a.test (HTTP 200)')
|
|
|
expect(out).toContain('# Title')
|
|
|
expect(out).toContain('Body text')
|
|
|
@@ -102,63 +113,149 @@ describe('fetch formatting', () => {
|
|
|
const out = formatFetchOutput({
|
|
|
url: 'https://a.test', statusCode: 200, truncated: true,
|
|
|
body: { kind: 'text', content: 'plain' },
|
|
|
- })
|
|
|
+ }, NO_CAP)
|
|
|
expect(out).toContain('plain')
|
|
|
expect(out).toContain('Content truncated')
|
|
|
})
|
|
|
|
|
|
- it('renderBody dispatches on kind', () => {
|
|
|
- expect(renderBody({ kind: 'text', content: 'x' })).toBe('x')
|
|
|
- expect(renderBody({ kind: 'html', content: '<p>y</p>' })).toBe('y')
|
|
|
+ it('caps the complete output and notes truncation, even when markdown escaping expands the body', () => {
|
|
|
+ // 1,000 underscores render as 2,000 escaped characters — conversion can
|
|
|
+ // outgrow a provider-side body cap, so the bound applies to the output.
|
|
|
+ const out = formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: `<p>${'_'.repeat(1000)}</p>` },
|
|
|
+ }, 500)
|
|
|
+ expect(out.length).toBeLessThanOrEqual(500)
|
|
|
+ expect(out).toContain('Fetched https://a.test (HTTP 200)')
|
|
|
+ expect(out).toContain('\\_\\_')
|
|
|
+ expect(out).toContain('Content truncated')
|
|
|
+ // Exact and tiny caps: the complete result is bounded, header and footer included.
|
|
|
+ const exact = formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'text', content: 'abc' },
|
|
|
+ }, 'Fetched https://a.test (HTTP 200)\n\nabc'.length)
|
|
|
+ expect(exact).toBe('Fetched https://a.test (HTTP 200)\n\nabc')
|
|
|
+ const tiny = formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: true,
|
|
|
+ body: { kind: 'text', content: 'abcdef' },
|
|
|
+ }, 10)
|
|
|
+ expect(tiny.length).toBeLessThanOrEqual(10)
|
|
|
+ expect(tiny).toBe('Fetched ht')
|
|
|
})
|
|
|
|
|
|
- it('validates url (non-empty), no timeout parameter', () => {
|
|
|
- expect(() => parseFetchArgs({ url: ' ' })).toThrow('non-empty')
|
|
|
- expect(parseFetchArgs({ url: 'https://a.test' })).toEqual({ url: 'https://a.test' })
|
|
|
+ it('dispatches text and html bodies', () => {
|
|
|
+ expect(formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'text', content: 'x' },
|
|
|
+ }, NO_CAP)).toBe(`${HEADER}x`)
|
|
|
+ expect(renderHtml('<p>y</p>')).toBe('y')
|
|
|
+ })
|
|
|
+
|
|
|
+ it('converts html via turndown: entities, links, tables, nesting; drops script/style/noscript', () => {
|
|
|
+ expect(renderHtml('<style>.x{}</style><script>bad()</script><noscript>ns</noscript><p>Tom & Jerry © Résumé</p><a href="https://a.test">link</a>'))
|
|
|
+ .toBe('Tom & Jerry © Résumé\n\n[link](https://a.test)')
|
|
|
+ expect(renderHtml('<h2>Heading</h2><ul><li>one</li><li>two</li></ul>'))
|
|
|
+ .toBe('## Heading\n\n- one\n- two')
|
|
|
+ expect(renderHtml('<table><tr><th>A</th><th>B</th></tr><tr><td>1</td><td>2</td></tr></table>'))
|
|
|
+ .toBe('| A | B |\n| --- | --- |\n| 1 | 2 |')
|
|
|
+ expect(renderHtml('<table><thead><tr><th align="left">L</th><th align="right">R</th><th style="text-align:center">C</th></tr></thead><tbody><tr><td>1</td><td>2</td><td>3</td></tr></tbody></table>'))
|
|
|
+ .toBe('| L | R | C |\n| :--- | ---: | :---: |\n| 1 | 2 | 3 |')
|
|
|
+ expect(renderHtml('<p><strong>bold <em>italic</em></strong></p><blockquote><p>quoted</p></blockquote>'))
|
|
|
+ .toBe('**bold _italic_**\n\n> quoted')
|
|
|
+ })
|
|
|
+
|
|
|
+ it('does not expand numeric colspan attributes into unbounded output', () => {
|
|
|
+ const table = '<table><thead><tr><th colspan="1000000">A</th></tr></thead><tbody><tr><td>B</td></tr></tbody></table>'
|
|
|
+ expect(renderHtml(table)).toBe('| A |\n| --- |\n| B |')
|
|
|
+ })
|
|
|
+
|
|
|
+ it('passes deeply nested html through raw without attempting conversion', () => {
|
|
|
+ // Unclosed-tag nesting makes the synchronous conversion superlinear
|
|
|
+ // (seconds at 20k levels, during which the cooperative timeout cannot
|
|
|
+ // fire), so the depth preflight skips conversion entirely; this must
|
|
|
+ // return fast, not merely not-throw.
|
|
|
+ const depth = 20_000
|
|
|
+ const pathological = '<div>'.repeat(depth) + 'x' + '</div>'.repeat(depth)
|
|
|
+ const started = Date.now()
|
|
|
+ expect(formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: pathological },
|
|
|
+ }, NO_CAP)).toBe(`${HEADER}${pathological}`)
|
|
|
+ expect(Date.now() - started).toBeLessThan(2_000)
|
|
|
})
|
|
|
|
|
|
- it('presents a fetch call as a fetch-kind card titled by the url', () => {
|
|
|
- expect(presentFetchCall({ url: 'https://a.test' })).toEqual({ card: 'generic', title: 'https://a.test', kind: 'fetch', rawInput: 'https://a.test' })
|
|
|
+ it('comments and mismatched closing tags cannot hide deep nesting from the preflight', () => {
|
|
|
+ const pathological = '<div><!-- </div> --></span>'.repeat(600) + 'x'
|
|
|
+ expect(formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: pathological },
|
|
|
+ }, NO_CAP)).toBe(`${HEADER}${pathological}`)
|
|
|
+ const abruptlyClosedComments = '<div><!-->'.repeat(600) + 'x'
|
|
|
+ expect(formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: abruptlyClosedComments },
|
|
|
+ }, NO_CAP)).toBe(`${HEADER}${abruptlyClosedComments}`)
|
|
|
})
|
|
|
-})
|
|
|
|
|
|
-describe('htmlToMarkdown', () => {
|
|
|
- it('drops scripts/styles, keeps text, decodes entities, converts links', () => {
|
|
|
- const md = htmlToMarkdown('<style>.x{}</style><script>bad()</script><p>Tom & Jerry</p><a href="https://a.test">link</a>')
|
|
|
- expect(md).not.toContain('bad()')
|
|
|
- expect(md).not.toContain('.x{}')
|
|
|
- expect(md).toContain('Tom & Jerry')
|
|
|
- expect(md).toContain('[link](https://a.test)')
|
|
|
+ it('the preflight accepts ordinary closed, void, self-closing, quoted, and raw-text markup', () => {
|
|
|
+ const paragraphs = '<p title=\'>\'>x<br ><img src="x"><input/></p>'.repeat(600)
|
|
|
+ const script = `<script>const invalid = '</scriptx>'; const template = '${'<div>'.repeat(600)}'</script >`
|
|
|
+ expect(renderHtml(`<!doctype html><?pi><1bad>${paragraphs}${script}`))
|
|
|
+ .not.toContain('<p')
|
|
|
+ expect(renderHtml('plain text')).toBe('plain text')
|
|
|
+ expect(renderHtml('<p>x</p><!-- unfinished')).toBe('x')
|
|
|
+ expect(renderHtml('<script>unclosed')).toBe('')
|
|
|
+ expect(renderHtml('<script>closed by slash</script/>')).toBe('')
|
|
|
+ expect(renderHtml('<script>closed at end</script')).toBe('')
|
|
|
})
|
|
|
|
|
|
- it('decodes numeric entities and collapses whitespace', () => {
|
|
|
- expect(htmlToMarkdown('<p>a'b</p>')).toBe("a'b")
|
|
|
- expect(htmlToMarkdown('<div>x</div>\n\n\n<div>y</div>')).toBe('x\n\ny')
|
|
|
+ it('scans malformed unterminated tags in bounded time', () => {
|
|
|
+ const malformed = '<a'.repeat(100_000)
|
|
|
+ const started = Date.now()
|
|
|
+ const out = formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: malformed },
|
|
|
+ }, 200_000)
|
|
|
+ expect(out.length).toBeLessThanOrEqual(200_000)
|
|
|
+ expect(Date.now() - started).toBeLessThan(2_000)
|
|
|
})
|
|
|
|
|
|
- it('decodes hex entities and named entities, and leaves unknown/out-of-range ones intact', () => {
|
|
|
- expect(htmlToMarkdown('<p>AB</p>')).toBe('AB')
|
|
|
- expect(htmlToMarkdown('<p>© —</p>')).toBe('© —')
|
|
|
- expect(htmlToMarkdown('<p>¬areal;</p>')).toBe('¬areal;')
|
|
|
- // An out-of-range code point keeps the original entity text (fromCodePoint fallback).
|
|
|
- expect(htmlToMarkdown('<p>�</p>')).toBe('�')
|
|
|
- expect(htmlToMarkdown('<p>�</p>')).toBe('�')
|
|
|
+ it('falls back to the raw html when turndown throws despite a shallow depth scan', () => {
|
|
|
+ const spy = vi.spyOn(TurndownService.prototype, 'turndown').mockImplementation(() => {
|
|
|
+ throw new RangeError('Maximum call stack size exceeded')
|
|
|
+ })
|
|
|
+ try {
|
|
|
+ expect(formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: '<p>x</p>' },
|
|
|
+ }, NO_CAP)).toBe(`${HEADER}<p>x</p>`)
|
|
|
+ } finally {
|
|
|
+ spy.mockRestore()
|
|
|
+ }
|
|
|
})
|
|
|
|
|
|
- it('renders a link with an empty label as its bare href', () => {
|
|
|
- expect(htmlToMarkdown('<a href="https://a.test"></a>')).toBe('https://a.test')
|
|
|
+ it('bounds source conversion work before rendering a custom provider body', () => {
|
|
|
+ const spy = vi.spyOn(TurndownService.prototype, 'turndown').mockReturnValue('converted')
|
|
|
+ try {
|
|
|
+ const out = formatFetchOutput({
|
|
|
+ url: 'https://a.test', statusCode: 200, truncated: false,
|
|
|
+ body: { kind: 'html', content: `<p>${'x'.repeat(10_000)}</p>` },
|
|
|
+ }, 500)
|
|
|
+ expect(spy).toHaveBeenCalledWith(`<p>${'x'.repeat(497)}`)
|
|
|
+ expect(out.length).toBeLessThanOrEqual(500)
|
|
|
+ expect(out).toContain('Content truncated')
|
|
|
+ } finally {
|
|
|
+ spy.mockRestore()
|
|
|
+ }
|
|
|
})
|
|
|
|
|
|
- it('converts headings and list items to markdown', () => {
|
|
|
- expect(htmlToMarkdown('<h2>Heading</h2><p>after</p>')).toContain('## Heading')
|
|
|
- const list = htmlToMarkdown('<ul><li>one</li><li>two</li></ul>')
|
|
|
- expect(list).toContain('- one')
|
|
|
- expect(list).toContain('- two')
|
|
|
+ it('validates url (non-empty), no timeout parameter', () => {
|
|
|
+ expect(() => parseFetchArgs({ url: ' ' })).toThrow('non-empty')
|
|
|
+ expect(parseFetchArgs({ url: 'https://a.test' })).toEqual({ url: 'https://a.test' })
|
|
|
})
|
|
|
|
|
|
- it('falls back to the raw URL as a source label when the URL is unparseable', () => {
|
|
|
- const out = formatSearchOutput({ truncated: false, sources: [{ url: 'not a url' }] })
|
|
|
- expect(out).toContain('[not a url](not a url)')
|
|
|
+ it('presents a fetch call as a fetch-kind card titled by the url', () => {
|
|
|
+ expect(presentFetchCall({ url: 'https://a.test' })).toEqual({ card: 'generic', title: 'https://a.test', kind: 'fetch', rawInput: 'https://a.test' })
|
|
|
})
|
|
|
})
|
|
|
|
|
|
@@ -395,3 +492,35 @@ describe('tool-call timeout budget is plugin config', () => {
|
|
|
.rejects.toThrow(new RegExp(`tool-web: ${key} must be a positive integer`))
|
|
|
})
|
|
|
})
|
|
|
+
|
|
|
+describe('fetchMaxOutputChars is plugin config', () => {
|
|
|
+ it('bounds the rendered output of the registered web_fetch tool', async () => {
|
|
|
+ const fetchProvider = {
|
|
|
+ id: 'stub-fetch',
|
|
|
+ available: () => available,
|
|
|
+ fetch: (request: { url: string }) => Promise.resolve({
|
|
|
+ url: request.url,
|
|
|
+ statusCode: 200,
|
|
|
+ body: { kind: 'html' as const, content: `<p>${'_'.repeat(1_000)}</p>` },
|
|
|
+ truncated: false,
|
|
|
+ }),
|
|
|
+ }
|
|
|
+ const { fiber, call } = await mountTools({
|
|
|
+ config: { fetchMaxOutputChars: 100 },
|
|
|
+ webConfig: { fetchProvider: 'stub-fetch' },
|
|
|
+ fetchProvider,
|
|
|
+ })
|
|
|
+ const out = await call('web_fetch', { url: 'https://a.test' })
|
|
|
+ expect(out.content.map(block => block.type === 'text' ? block.text : '').join('')).toHaveLength(100)
|
|
|
+ await fiber.dispose()
|
|
|
+ })
|
|
|
+
|
|
|
+ it.each([0, -1, 1.5])('rejects an invalid fetchMaxOutputChars value %s at load', async (value) => {
|
|
|
+ const ctx = new Context()
|
|
|
+ await ctx.plugin(SystemPrompt)
|
|
|
+ await ctx.plugin(ToolRegistry)
|
|
|
+ await ctx.plugin(WebService, {})
|
|
|
+ await expect(ctx.plugin(ToolWeb, { fetchMaxOutputChars: value }))
|
|
|
+ .rejects.toThrow(/tool-web: fetchMaxOutputChars must be a positive integer/)
|
|
|
+ })
|
|
|
+})
|