|
|
@@ -1,16 +1,19 @@
|
|
|
/**
|
|
|
* Shared execution plumbing for the `glob` / `grep` search tools: the
|
|
|
- * package-owned `SEARCH_*` error vocabulary, one bash-seam run helper that
|
|
|
- * turns a fixed `rg` command into complete raw stdout, the best-effort
|
|
|
- * formatted-result spill handoff, and workdir-relative path display.
|
|
|
+ * package-owned `SEARCH_*` error vocabulary, one spawn helper that runs the
|
|
|
+ * PACKAGED ripgrep binary (`@vscode/ripgrep`) with a plain argv vector and
|
|
|
+ * returns complete raw stdout, the best-effort formatted-result spill handoff,
|
|
|
+ * and workdir-relative path display.
|
|
|
*
|
|
|
- * Both tools execute through `ctx.bash.resolve(request)` → `ctx.bash.run(spec)`
|
|
|
- * as ordinary foreground tool calls — never `ctx.bash.start()`, never a
|
|
|
- * model-visible background task. Raw `rg` stdout is an internal transport
|
|
|
- * detail: the tools request a per-run stdout capture budget from the bash seam,
|
|
|
- * parse only complete in-memory stdout within `rawOutputMaxBytes`, and never
|
|
|
- * read executor spill files. The model-facing recovery artifact is the
|
|
|
- * formatted result saved through `ctx.spillStore.saveText()`
|
|
|
+ * Both tools execute as ordinary foreground spawns through `ctx.subprocess` —
|
|
|
+ * never `ctx.bash`, never `ctx.bash.start()`, never a model-visible background
|
|
|
+ * task. The ripgrep binary ships inside the npm package, so no system `rg`
|
|
|
+ * install is required, and no shell layer exists between the argv vector and
|
|
|
+ * ripgrep, so no shell quoting is involved. Raw `rg` stdout is an internal
|
|
|
+ * transport detail: the tools request a per-run stdout capture budget from the
|
|
|
+ * subprocess seam, parse only complete in-memory stdout within
|
|
|
+ * `rawOutputMaxBytes`, and never read spill files. The model-facing recovery
|
|
|
+ * artifact is the formatted result saved through `ctx.spillStore.saveText()`
|
|
|
* ({@link trySaveFormattedResult}).
|
|
|
*
|
|
|
* @module @deepseek-ai/dsh-tool-fs-search/search-core
|
|
|
@@ -18,10 +21,11 @@
|
|
|
|
|
|
import { isAbsolute, relative, sep } from 'node:path'
|
|
|
import type { Context } from 'cordis'
|
|
|
+import { rgPath } from '@vscode/ripgrep'
|
|
|
import { HarnessError } from '@deepseek-ai/dsh-llm'
|
|
|
import { ItemRetainer, TextRetainer } from '@deepseek-ai/dsh-retention'
|
|
|
import type { RetainedItems } from '@deepseek-ai/dsh-retention'
|
|
|
-import type { BashRunResult, CollectedOutput } from '@deepseek-ai/dsh-bash'
|
|
|
+import type { SubprocessCollect, SubprocessOutcome, SubprocessOutputRead, SubprocessSpawnSpec } from '@deepseek-ai/dsh-subprocess'
|
|
|
import type { SaveTextSpill, SpillRef } from '@deepseek-ai/dsh-spill'
|
|
|
import type { ToolExecution } from '@deepseek-ai/dsh-tools'
|
|
|
|
|
|
@@ -38,6 +42,18 @@ export const RAW_OUTPUT_MAX_BYTES = 20_000_000
|
|
|
*/
|
|
|
export const SEARCH_TIMEOUT_MS = 30_000
|
|
|
|
|
|
+/**
|
|
|
+ * Default cap in bytes on the retained stderr tail of one search run — a
|
|
|
+ * diagnostic excerpt only (the tool never reads `stderr.spillPath`).
|
|
|
+ */
|
|
|
+const SEARCH_STDERR_MAX_BYTES = 64 * 1024
|
|
|
+
|
|
|
+/** Default whole-stream spill cap for search output (the subprocess seam requires an explicit budget). */
|
|
|
+const SEARCH_SPILL_MAX_BYTES = 64 * 1024 * 1024
|
|
|
+
|
|
|
+/** Default terminate grace period for a search process (ms). */
|
|
|
+const SEARCH_GRACE_MS = 3_000
|
|
|
+
|
|
|
/**
|
|
|
* Default cap in bytes on one search's serialized `presentationMeta` (the
|
|
|
* `searchMetaMaxBytes` config). The inline match/path caps already bound the item
|
|
|
@@ -52,14 +68,14 @@ export const SEARCH_META_MAX_BYTES = 65_536
|
|
|
|
|
|
/**
|
|
|
* Stable, machine-routable codes for search failures. Package-owned (not
|
|
|
- * `FsErrorCode`) because these tools are bash-backed discovery, not `ctx.fs`
|
|
|
+ * `FsErrorCode`) because these tools are spawn-backed discovery, not `ctx.fs`
|
|
|
* provider operations: `SEARCH_INVALID_PATTERN` — ripgrep rejected the regex or
|
|
|
* glob; `SEARCH_FAILED` — the search could not run or its output could not be
|
|
|
- * parsed (missing `rg`, inaccessible target, signal kill, malformed `--json`);
|
|
|
- * `SEARCH_RAW_OUTPUT_OVERFLOW` — raw `rg` output exceeded `rawOutputMaxBytes`
|
|
|
- * or stayed truncated after that requested stdout budget; `SEARCH_ABORTED` — the tool
|
|
|
- * timeout, caller cancellation, or the bash executor's own timeout cut the
|
|
|
- * search short.
|
|
|
+ * parsed (a failed `rg` launch, inaccessible target, signal kill, malformed
|
|
|
+ * `--json`); `SEARCH_RAW_OUTPUT_OVERFLOW` — raw `rg` output exceeded
|
|
|
+ * `rawOutputMaxBytes` or stayed truncated after that requested stdout budget;
|
|
|
+ * `SEARCH_ABORTED` — the cooperative tool timeout or caller cancellation cut
|
|
|
+ * the search short.
|
|
|
*/
|
|
|
export type SearchErrorCode =
|
|
|
| 'SEARCH_INVALID_PATTERN'
|
|
|
@@ -84,7 +100,7 @@ export class SearchError extends HarnessError {
|
|
|
|
|
|
/** The completed acquisition of one `rg` run: complete stdout plus the resolved workdir. */
|
|
|
export interface RipgrepRun {
|
|
|
- /** Complete raw stdout retained by the bash executor within the requested cap. */
|
|
|
+ /** Complete raw stdout retained by the subprocess seam within the requested cap. */
|
|
|
stdout: string
|
|
|
/** True when ripgrep exited 1: a successful search with zero results. */
|
|
|
noMatches: boolean
|
|
|
@@ -94,73 +110,71 @@ export interface RipgrepRun {
|
|
|
|
|
|
/**
|
|
|
* The retained stderr tail as a diagnostic excerpt, with a truncation note when
|
|
|
- * the executor dropped bytes (the tool never reads `stderr.spillPath`).
|
|
|
+ * the subprocess seam dropped bytes (the tool never reads `stderr.spillPath`).
|
|
|
*/
|
|
|
-function stderrExcerpt(stderr: CollectedOutput): string {
|
|
|
- const text = stderr.text.trim()
|
|
|
+function stderrExcerpt(stderrText: string, truncated: boolean): string {
|
|
|
+ const text = stderrText.trim()
|
|
|
if (text.length === 0) return ''
|
|
|
- return stderr.truncated ? `${text} [stderr truncated]` : text
|
|
|
+ return truncated ? `${text} [stderr truncated]` : text
|
|
|
}
|
|
|
|
|
|
/** Classify a nonzero-exit `rg` run into the search error vocabulary (invalid pattern vs missing `rg` vs everything else). */
|
|
|
-function classifyRunFailure(toolName: string, result: BashRunResult): SearchError {
|
|
|
- const stderr = stderrExcerpt(result.stderr)
|
|
|
+function classifyRunFailure(toolName: string, exitCode: number, stderrText: string, stderrTruncated: boolean): SearchError {
|
|
|
+ const stderr = stderrExcerpt(stderrText, stderrTruncated)
|
|
|
if (/regex parse error|error parsing glob/i.test(stderr)) {
|
|
|
return new SearchError(`${toolName} pattern rejected by ripgrep: ${stderr}`, 'SEARCH_INVALID_PATTERN')
|
|
|
}
|
|
|
- if (result.exitCode === 127 || /command not found/i.test(stderr)) {
|
|
|
- return new SearchError(`${toolName} requires ripgrep (rg) on the bash executor's PATH${stderr.length > 0 ? `: ${stderr}` : ''}`, 'SEARCH_FAILED')
|
|
|
+ if (exitCode === 127 || /command not found/i.test(stderr)) {
|
|
|
+ return new SearchError(`${toolName} requires ripgrep (rg) to launch${stderr.length > 0 ? `: ${stderr}` : ''}`, 'SEARCH_FAILED')
|
|
|
}
|
|
|
- return new SearchError(`${toolName} search failed (exit ${result.exitCode})${stderr.length > 0 ? `: ${stderr}` : ''}`, 'SEARCH_FAILED')
|
|
|
+ return new SearchError(`${toolName} search failed (exit ${exitCode})${stderr.length > 0 ? `: ${stderr}` : ''}`, 'SEARCH_FAILED')
|
|
|
}
|
|
|
|
|
|
/**
|
|
|
* Acquire the COMPLETE raw stdout of a finished run, enforcing
|
|
|
* `rawOutputMaxBytes` on the in-memory transport. A truncated result means the
|
|
|
- * bash backend could not retain complete stdout within the requested budget, so
|
|
|
- * the tool fails clearly instead of parsing a silently-partial stream.
|
|
|
+ * subprocess seam could not retain complete stdout within the requested
|
|
|
+ * budget, so the tool fails clearly instead of parsing a silently-partial
|
|
|
+ * stream.
|
|
|
*/
|
|
|
-function completeStdout(toolName: string, result: BashRunResult, rawOutputMaxBytes: number): string {
|
|
|
+function completeStdout(toolName: string, stdout: SubprocessOutputRead, rawOutputMaxBytes: number): string {
|
|
|
const narrow = 'narrow pattern, path, or include and retry'
|
|
|
- if (!result.stdout.truncated) {
|
|
|
- const inlineBytes = Buffer.byteLength(result.stdout.text, 'utf8')
|
|
|
+ if (!stdout.lossy) {
|
|
|
+ const inlineBytes = Buffer.byteLength(stdout.text, 'utf8')
|
|
|
if (inlineBytes > rawOutputMaxBytes) {
|
|
|
throw new SearchError(
|
|
|
`${toolName} produced ${inlineBytes} bytes of raw output, over the ${rawOutputMaxBytes}-byte cap; ${narrow}`,
|
|
|
'SEARCH_RAW_OUTPUT_OVERFLOW',
|
|
|
)
|
|
|
}
|
|
|
- return result.stdout.text
|
|
|
+ return stdout.text
|
|
|
}
|
|
|
throw new SearchError(
|
|
|
- `${toolName} produced more raw output than the bash executor retained within the ${rawOutputMaxBytes}-byte cap; ${narrow}`,
|
|
|
+ `${toolName} produced more raw output than the subprocess seam retained within the ${rawOutputMaxBytes}-byte cap; ${narrow}`,
|
|
|
'SEARCH_RAW_OUTPUT_OVERFLOW',
|
|
|
)
|
|
|
}
|
|
|
|
|
|
/**
|
|
|
- * Run one fixed `rg` command through the bash seam and return its complete raw
|
|
|
- * stdout. The bash request workdir is the calling agent's session cwd
|
|
|
- * (`exec.agent.session.header.cwd`) when available — mirroring `dsh-tool-bash` /
|
|
|
- * `dsh-tool-fs` — else omitted so the implementation's `resolve()` applies its
|
|
|
- * configured default. `exec.signal` is forwarded so the cooperative tool
|
|
|
- * timeout (`@deepseek-ai/dsh-timeout-policy`) and caller cancellation kill the
|
|
|
- * command; the bash backend's own timeout stays a second safety cap.
|
|
|
+ * Run the packaged ripgrep binary with a plain argv vector and return its
|
|
|
+ * complete raw stdout. The working directory is the calling agent's session
|
|
|
+ * cwd (`exec.agent.session.header.cwd`) when available, else
|
|
|
+ * `process.cwd()`. `exec.signal` is forwarded so the cooperative tool timeout
|
|
|
+ * (`@deepseek-ai/dsh-timeout-policy`) and caller cancellation terminate the
|
|
|
+ * process tree.
|
|
|
*
|
|
|
* Exit semantics are tool-owned: exit 0 is success with results, exit 1 is
|
|
|
* success with zero results (`noMatches`), anything else throws a
|
|
|
* {@link SearchError} (abort/timeout → `SEARCH_ABORTED`, invalid pattern →
|
|
|
* `SEARCH_INVALID_PATTERN`, the rest → `SEARCH_FAILED` /
|
|
|
- * `SEARCH_RAW_OUTPUT_OVERFLOW`). A `run()` REJECTION — the seam's
|
|
|
- * infrastructure failures (pre-aborted signal, unusable workdir, missing
|
|
|
- * shell) — is translated into the same taxonomy: a pre-aborted signal becomes
|
|
|
- * `SEARCH_ABORTED`, everything else `SEARCH_FAILED`, with the original as
|
|
|
- * `cause`.
|
|
|
+ * `SEARCH_RAW_OUTPUT_OVERFLOW`). A spawn REJECTION — the seam's
|
|
|
+ * infrastructure failures — is translated into `SEARCH_FAILED` with the
|
|
|
+ * original as `cause`; a pre-aborted signal becomes `SEARCH_ABORTED`.
|
|
|
*
|
|
|
- * @param ctx - the plugin context; execution uses its `bash` service.
|
|
|
+ * @param ctx - the plugin context; execution uses its `subprocess` service.
|
|
|
* @param exec - the tool-execution context; supplies the session cwd and the abort signal.
|
|
|
* @param toolName - `glob` or `grep`, used in error messages.
|
|
|
- * @param command - the fully-quoted `rg` command string (every model value already through `singleQuote`).
|
|
|
+ * @param argv - the ripgrep arguments (every model value an unquoted argv element; no shell layer exists).
|
|
|
* @param rawOutputMaxBytes - cap on the complete raw stdout the tool will parse.
|
|
|
* @returns the complete stdout, the zero-result flag, and the resolved workdir.
|
|
|
*/
|
|
|
@@ -168,54 +182,64 @@ export async function runRipgrep(
|
|
|
ctx: Context,
|
|
|
exec: ToolExecution,
|
|
|
toolName: string,
|
|
|
- command: string,
|
|
|
+ argv: readonly string[],
|
|
|
rawOutputMaxBytes: number,
|
|
|
): Promise<RipgrepRun> {
|
|
|
+ if (exec.signal.aborted) {
|
|
|
+ throw new SearchError(`${toolName} was aborted before completion (tool timeout or caller cancellation)`, 'SEARCH_ABORTED')
|
|
|
+ }
|
|
|
const cwd = exec.agent?.session.header.cwd
|
|
|
- const spec = ctx.bash.resolve({
|
|
|
- command,
|
|
|
- stdoutMaxBytes: rawOutputMaxBytes,
|
|
|
- ...cwd !== undefined ? { workdir: cwd } : {},
|
|
|
+ const workdir = cwd ?? process.cwd()
|
|
|
+ const collect = (maxBytes: number): SubprocessCollect =>
|
|
|
+ ({ maxBytes, spill: { maxBytes: SEARCH_SPILL_MAX_BYTES } })
|
|
|
+ const handle = ctx.subprocess.spawn({
|
|
|
+ argv: [rgPath, ...argv],
|
|
|
+ cwd: workdir,
|
|
|
+ stdio: {
|
|
|
+ stdin: 'ignore',
|
|
|
+ stdout: collect(rawOutputMaxBytes),
|
|
|
+ stderr: collect(SEARCH_STDERR_MAX_BYTES),
|
|
|
+ },
|
|
|
+ graceMs: SEARCH_GRACE_MS,
|
|
|
signal: exec.signal,
|
|
|
- })
|
|
|
- let result: BashRunResult
|
|
|
+ } satisfies SubprocessSpawnSpec)
|
|
|
+ let outcome: SubprocessOutcome
|
|
|
try {
|
|
|
- result = await ctx.bash.run(spec)
|
|
|
+ outcome = await handle.done
|
|
|
} catch (error: unknown) {
|
|
|
- // The seam contract: run() REJECTS only for infrastructure failures — a
|
|
|
- // pre-aborted signal, an unusable workdir, a missing shell. Translate them
|
|
|
- // so these failures stay machine-routable under the SEARCH_* taxonomy.
|
|
|
- if (spec.signal?.aborted === true) {
|
|
|
- throw new SearchError(`${toolName} was aborted before completion (tool timeout or caller cancellation)`, 'SEARCH_ABORTED', { cause: error })
|
|
|
- }
|
|
|
- throw new SearchError(`${toolName} could not start its search command (unusable working directory or missing shell)`, 'SEARCH_FAILED', { cause: error })
|
|
|
+ throw new SearchError(`${toolName} could not start its search command (ripgrep launch failed)`, 'SEARCH_FAILED', { cause: error })
|
|
|
}
|
|
|
- if (result.aborted) {
|
|
|
- throw new SearchError(`${toolName} was aborted before completion (tool timeout or caller cancellation)`, 'SEARCH_ABORTED')
|
|
|
+ const stdout = handle.collected.stdout?.readFrom(0)
|
|
|
+ const stderr = handle.collected.stderr?.readFrom(0)
|
|
|
+ if (stdout === undefined || stderr === undefined) {
|
|
|
+ throw new SearchError(`${toolName} search command produced no collected output streams`, 'SEARCH_FAILED')
|
|
|
}
|
|
|
- if (result.timedOut) {
|
|
|
- throw new SearchError(`${toolName} timed out after ${result.timeoutMs}ms in the bash executor; narrow pattern, path, or include and retry`, 'SEARCH_ABORTED')
|
|
|
+ // The signal can abort while the spawn is awaited; the static narrowing that
|
|
|
+ // proves this re-check "always false" cannot see AbortSignal state changes.
|
|
|
+ // oxlint-disable-next-line typescript/no-unnecessary-condition
|
|
|
+ if (exec.signal.aborted) {
|
|
|
+ throw new SearchError(`${toolName} was aborted before completion (tool timeout or caller cancellation)`, 'SEARCH_ABORTED')
|
|
|
}
|
|
|
- if (result.signal !== null || result.exitCode === null) {
|
|
|
- throw new SearchError(`${toolName} search command was killed by signal ${result.signal ?? '(unknown)'}`, 'SEARCH_FAILED')
|
|
|
+ if (outcome.signal !== null || outcome.exitCode === null) {
|
|
|
+ throw new SearchError(`${toolName} search command was killed by signal ${outcome.signal ?? '(unknown)'}`, 'SEARCH_FAILED')
|
|
|
}
|
|
|
- if (result.exitCode !== 0 && result.exitCode !== 1) {
|
|
|
- throw classifyRunFailure(toolName, result)
|
|
|
+ if (outcome.exitCode !== 0 && outcome.exitCode !== 1) {
|
|
|
+ throw classifyRunFailure(toolName, outcome.exitCode, stderr.text, stderr.lossy)
|
|
|
}
|
|
|
- const stdout = completeStdout(toolName, result, rawOutputMaxBytes)
|
|
|
- return { stdout, noMatches: result.exitCode === 1, workdir: spec.workdir }
|
|
|
+ const text = completeStdout(toolName, stdout, rawOutputMaxBytes)
|
|
|
+ return { stdout: text, noMatches: outcome.exitCode === 1, workdir }
|
|
|
}
|
|
|
|
|
|
/**
|
|
|
* Map an `rg` output path to its display form: absolute paths inside the
|
|
|
- * resolved bash workdir become workdir-relative; everything else (relative
|
|
|
- * output, paths outside the workdir) passes through unchanged. Display-only —
|
|
|
- * returned paths are follow-up-readable in co-located bash/filesystem
|
|
|
+ * resolved workdir become workdir-relative; everything else (relative output,
|
|
|
+ * paths outside the workdir) passes through unchanged. Display-only —
|
|
|
+ * returned paths are follow-up-readable in co-located workdir/filesystem
|
|
|
* deployments where both resolve the same workspace (the documented v1
|
|
|
* deployment requirement).
|
|
|
*
|
|
|
* @param path - one path as ripgrep printed it.
|
|
|
- * @param workdir - the resolved bash workdir the command ran in.
|
|
|
+ * @param workdir - the resolved workdir the command ran in.
|
|
|
* @returns the workdir-relative display path when possible, else `path` unchanged.
|
|
|
*/
|
|
|
export function toWorkdirRelative(path: string, workdir: string): string {
|