index.ts 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204
  1. /**
  2. * The model-facing `workflow` tool: run a JavaScript orchestration script that fans out
  3. * subagents, and return the script's final value. Pure schema + lifecycle shaping — script
  4. * parsing, execution, caps, and cancellation live behind `ctx.workflows`
  5. * (`@deepseek-ai/dsh-workflow`), so a hardened engine swaps in without touching what the model
  6. * sees. Execution awaits `run.result` and always disposes the run; non-completed reasons become tool
  7. * errors, and background collection remains deferred. Presentation is an args-only generic card
  8. * titled from `meta.name`. Explicit-ask usage guidance is registered as the tool's own prompt
  9. * section rather than deployment persona prose.
  10. * @module @deepseek-ai/dsh-tool-workflow
  11. */
  12. import type { Context } from 'cordis'
  13. import z from 'schemastery'
  14. import { defineTool } from '@deepseek-ai/dsh-tools'
  15. import type { ToolCallView, ToolResultView } from '@deepseek-ai/dsh-tools'
  16. import type { ContentBlock } from '@deepseek-ai/dsh-llm'
  17. import type { WorkflowResult, WorkflowRun } from '@deepseek-ai/dsh-workflow'
  18. // Declaration merge only: makes ctx.systemPrompt visible for the section registration.
  19. import type {} from '@deepseek-ai/dsh-system-prompt'
  20. export const name = 'tool-workflow'
  21. export const inject = ['tools', 'workflows', 'systemPrompt']
  22. /** Config: the model-facing tool name plus result rendering caps. */
  23. export interface Config {
  24. /** The model-facing tool name to register (default `workflow`). */
  25. toolName?: string
  26. /** Rendered-result ceiling, in characters: a longer JSON value is truncated with a notice (default 50000). */
  27. maxResultChars?: number
  28. }
  29. export const Config: z<Config> = z.object({
  30. toolName: z.string().default('workflow'),
  31. maxResultChars: z.natural().min(1).default(50_000),
  32. })
  33. type ResolvedConfig = Required<Config>
  34. /**
  35. * The script-authoring contract, embedded in the tool description. This IS the
  36. * model-facing spec: the meta block, the hooks and their exact semantics, and
  37. * the supported schema subset.
  38. */
  39. const DESCRIPTION = `Run a JavaScript workflow script that orchestrates subagents at scale. Use this for work that fans out across many independent pieces — an audit over many files, a migration, multi-angle research, adversarial verification of findings — where you write the orchestration as a script instead of delegating turn by turn.
  40. The workflow's identity rides the \`meta\` parameter as JSON: required \`name\` (short kebab-case) and \`description\` strings, optional \`whenToUse\` string and \`phases\` array (\`{title, detail?, provider?, model?}\`). The \`script\` parameter is the plain JavaScript body ONLY (NOT TypeScript, and NO \`export const meta\` statement — meta is a parameter, not code), running with top-level await; end with \`return <value>\` — the value must be JSON-serializable and is this tool's result.
  41. Script-body hooks:
  42. - \`agent(prompt, opts?): Promise<any>\` — run one subagent to completion. Without \`opts.schema\` it resolves to the child's final text; with \`opts.schema\` (an object-rooted JSON Schema using ONLY type/properties/required/additionalProperties/items/enum/const — no oneOf/pattern/format/numeric bounds) it resolves to the validated object. Resolves \`null\` when the child fails (filter with \`.filter(Boolean)\`). Other opts: \`label\` (display), \`phase\` (progress group), and independent \`provider\`/\`model\` LLM target overrides (either may be provided alone). Anything else (\`effort\`/\`isolation\`/\`agentType\`) is rejected loudly.
  43. - \`pipeline(items, ...stages): Promise<any[]>\` — run each item through the stages independently with NO barrier between stages (prefer this for multi-stage work). Each stage receives \`(prev, item, index)\`. An ordinary stage throw drops that ITEM to \`null\` and skips its remaining stages.
  44. - \`parallel(thunks): Promise<any[]>\` — run zero-argument functions concurrently and await ALL of them (a barrier; use only when a stage genuinely needs every prior result together). A throwing thunk resolves to \`null\`.
  45. - \`phase(title)\` — start a progress phase; \`log(message)\` — narrate progress; \`args\` — the tool call's \`args\` input, verbatim.
  46. Misused hooks (bad arguments, unknown options, unsupported schemas, tripped caps) throw errors that ALWAYS kill the script — they never dissolve into a per-item \`null\`.
  47. Constraints: concurrency and total-agent caps apply; no filesystem, network, timers, or Node.js APIs are provided — the agents do the work, the script only coordinates them. The run executes in the foreground: this call returns when the whole script finishes.`
  48. type WorkflowCallArgs = {
  49. script: string
  50. meta: {
  51. name: string
  52. description: string
  53. whenToUse?: string
  54. phases?: { title: string; detail?: string; provider?: string; model?: string }[]
  55. }
  56. args?: Record<string, unknown>
  57. }
  58. /** The pending-state card: a generic card titled by the workflow's meta name. */
  59. function presentWorkflowCall(args: WorkflowCallArgs): ToolCallView {
  60. return {
  61. card: 'generic',
  62. title: `workflow: ${args.meta.name}`,
  63. rawInput: args.script,
  64. }
  65. }
  66. /** The completed-state card: keep the pending title; render the result content as-is. */
  67. function presentWorkflowResult(args: WorkflowCallArgs, result: { content: ContentBlock[]; isError: boolean }): ToolResultView {
  68. void args
  69. void result
  70. return { card: 'generic' }
  71. }
  72. /** A non-`completed` stop reason means the script did not finish cleanly. */
  73. function stopReasonError(result: WorkflowResult): string | undefined {
  74. switch (result.stopReason) {
  75. case 'completed':
  76. return undefined
  77. case 'cancelled':
  78. return `workflow run was cancelled${result.error !== undefined ? ` (${result.error})` : ''}`
  79. case 'error':
  80. return `workflow run failed: ${result.error ?? 'unknown error'}`
  81. /* v8 ignore start -- defensive: WorkflowStopReason is a closed union, exhaustive by construction; a future variant fails here loudly */
  82. default:
  83. return `workflow run ended abnormally (${String(result.stopReason satisfies never)})`
  84. /* v8 ignore stop */
  85. }
  86. }
  87. /** Render the run's outcome text: the meta name, agent count, and the JSON value (capped). */
  88. function renderResult(run: WorkflowRun, result: WorkflowResult, maxChars: number): string {
  89. // The engine returns JSON data (null for a valueless script), so stringify never yields undefined.
  90. const rendered = JSON.stringify(result.value, null, 2)
  91. const clipped = rendered.length > maxChars
  92. ? `${rendered.slice(0, maxChars)}\n… [truncated: ${rendered.length - maxChars} more characters]`
  93. : rendered
  94. return `workflow "${run.meta.name}" completed (${result.agentsStarted} agent${result.agentsStarted === 1 ? '' : 's'}).\nReturn value:\n${clipped}`
  95. }
  96. export function apply(ctx: Context, config: Config): void {
  97. // schemastery (the exported Config schema) has already filled the defaulted
  98. // fields; the assertion records that resolution, not a hidden fallback.
  99. const { toolName, maxResultChars } = config as ResolvedConfig
  100. // Usage policy ships with the tool (the master convention: tool guidance
  101. // lives in tool plugins as prompt sections, not in the deployment persona).
  102. ctx.systemPrompt.section({
  103. name: `tool:${toolName}`,
  104. order: 115,
  105. text: `Use the ${toolName} tool ONLY when the user explicitly asks for a workflow or for large multi-agent orchestration: you write a JavaScript script (the tool description documents the exact format) that fans work out across many subagents with phases and structured results. For one or two delegations, prefer plain subagent calls.`,
  106. })
  107. ctx.tools.register(defineTool({
  108. name: toolName,
  109. description: DESCRIPTION,
  110. parameters: {
  111. script: {
  112. type: 'string',
  113. required: true,
  114. description: 'The plain-JS workflow script body (top-level await allowed; NO `export const meta` statement; end with `return <json-value>`).',
  115. },
  116. meta: {
  117. type: 'object',
  118. required: true,
  119. description: 'The workflow identity block (plain JSON — never code).',
  120. properties: {
  121. name: { type: 'string', required: true, description: 'Short kebab-case workflow name.' },
  122. description: { type: 'string', required: true, description: 'One-line description of what the workflow does.' },
  123. whenToUse: { type: 'string', description: 'Optional guidance on when this workflow applies.' },
  124. phases: {
  125. type: 'array',
  126. description: 'Optional phase declarations matched by phase() calls.',
  127. items: {
  128. type: 'object',
  129. properties: {
  130. title: { type: 'string', required: true, description: 'The phase title phase() calls match by exact string.' },
  131. detail: { type: 'string', description: 'Optional one-line description of the phase.' },
  132. provider: { type: 'string', description: 'Optional provider override this phase is expected to use.' },
  133. model: { type: 'string', description: 'Optional model override this phase is expected to use.' },
  134. },
  135. },
  136. },
  137. },
  138. },
  139. args: {
  140. type: 'object',
  141. description: 'Optional JSON input exposed to the script as the `args` global (wrap a bare list as a field, e.g. {"files": [...]}).',
  142. },
  143. },
  144. async execute(args, exec): Promise<ContentBlock[]> {
  145. const parent = exec.agent
  146. if (!parent) {
  147. // The loop sets `exec.agent` for every model-driven call; its absence
  148. // means a non-agent caller invoked the tool directly, which has no
  149. // parent to attribute the children to. Fail loud rather than guess.
  150. throw new Error('workflow tool requires a calling agent (exec.agent was undefined)')
  151. }
  152. // Meta/body validation failures (META_INVALID/SCRIPT_PARSE) throw
  153. // synchronously here and become isError results via the registry — the
  154. // model sees the violation list and can correct the call.
  155. const run: WorkflowRun = ctx.workflows.start({
  156. script: args.script,
  157. meta: args.meta,
  158. ...args.args !== undefined ? { args: args.args } : {},
  159. parent,
  160. signal: exec.signal,
  161. })
  162. // Bridge the tool's abort signal to the run: if the parent step is aborted while the
  163. // script is in flight, cancel the whole run. The signal also enters the engine directly, but
  164. // this local bridge preserves the tool contract even if an implementation ignores it.
  165. const onAbort = (): void => { run.cancel('parent step aborted') }
  166. exec.signal.addEventListener('abort', onAbort, { once: true })
  167. try {
  168. const result = await run.result
  169. const error = stopReasonError(result)
  170. if (error !== undefined) {
  171. // Map a non-clean finish to an isError result (the registry turns a
  172. // throw into an isError). Report the reason, not partial output.
  173. throw new Error(error)
  174. }
  175. return [{ type: 'text', text: renderResult(run, result, maxResultChars) }]
  176. } finally {
  177. exec.signal.removeEventListener('abort', onAbort)
  178. // Always reach run quiescence — never leak a live script or children.
  179. await run.dispose()
  180. }
  181. },
  182. presentCall: args => presentWorkflowCall(args),
  183. presentResult: (args, result) => presentWorkflowResult(args, result),
  184. }))
  185. }