Просмотр исходного кода

Merge remote-tracking branch 'origin/master' into compact-basic-refactor

# Conflicts:
#	docs/cordis-catalog/events-and-services.md
#	examples/AGENTS.md
#	examples/coding-agent/cordis.yml
#	examples/coding-agent/tests/harness.ts
#	scripts/gen-cordis-catalog.ts
Hypatia May 2 месяцев назад
Родитель
Сommit
0c4059fc84
44 измененных файлов с 1169 добавлено и 23 удалено
  1. 14 0
      AGENTS.md
  2. 2 1
      docs/architecture.md
  3. 1 1
      docs/cookbook/adding-a-package.md
  4. 2 2
      docs/cordis-catalog/events-and-services.md
  5. 1 1
      docs/core-data-structures/core.md
  6. 25 0
      docs/core-data-structures/session.md
  7. 4 0
      docs/module-graph.md
  8. 1 0
      docs/rfc/README.md
  9. 58 0
      docs/rfc/implemented/feature/2026-06-29-todo-write-tool.md
  10. 1 1
      examples/AGENTS.md
  11. 1 1
      examples/README.md
  12. 14 1
      examples/acp-agent/cordis.snapshot.yml
  13. 14 1
      examples/acp-agent/cordis.yml
  14. 1 0
      examples/acp-agent/tests/acp.snapshot.ts
  15. 7 0
      examples/acp-agent/tests/snapshots/todo-plan/input.json
  16. 124 0
      examples/acp-agent/tests/snapshots/todo-plan/session.jsonl
  17. 51 0
      examples/acp-agent/tests/snapshots/todo-plan/stdout.golden.jsonl
  18. 1 1
      examples/coding-agent/README.md
  19. 15 2
      examples/coding-agent/cordis.yml
  20. 13 4
      examples/coding-agent/tests/harness.ts
  21. 49 0
      examples/coding-agent/tests/todo-write.e2e.ts
  22. 3 0
      packages/README.md
  23. 32 0
      packages/core/session/src/types.ts
  24. 61 1
      packages/core/session/tests/session.spec.ts
  25. 7 0
      packages/support/ui-stdio/src/index.ts
  26. 29 0
      packages/support/ui-stdio/tests/ui-stdio.spec.ts
  27. 9 0
      packages/todo/README.md
  28. 25 0
      packages/todo/tool-todo/README.md
  29. 39 0
      packages/todo/tool-todo/package.json
  30. 123 0
      packages/todo/tool-todo/src/index.ts
  31. 107 0
      packages/todo/tool-todo/tests/integration.spec.ts
  32. 167 0
      packages/todo/tool-todo/tests/tool-todo.spec.ts
  33. 27 0
      packages/todo/tool-todo/tsconfig.json
  34. 1 0
      packages/ui/acp/package.json
  35. 19 1
      packages/ui/acp/src/index.ts
  36. 10 0
      packages/ui/acp/tests/harness.ts
  37. 38 0
      packages/ui/acp/tests/load.spec.ts
  38. 38 1
      packages/ui/acp/tests/stream-update.spec.ts
  39. 27 0
      pnpm-lock.yaml
  40. 2 2
      scripts/gen-cordis-catalog.ts
  41. 1 0
      scripts/type-equiv.manifest.json
  42. 1 0
      tsconfig.base.json
  43. 2 1
      tsconfig.build.json
  44. 2 1
      tsconfig.json

+ 14 - 0
AGENTS.md

@@ -73,6 +73,10 @@ packages/    Harness packages, grouped by role at packages/<group>/<pkg>/.
     bash/           abstract bash executor seam (ctx.bash) — interface only
     bash-local/     local-subprocess BashExecutor implementation
     tool-bash/      model-facing bash/bash_output/bash_kill tool schemas
+  todo/           todo/planning capability family
+    tool-todo/      model-facing todo_write tool: writes the whole task list to
+                    the session log (todo/write), rendered as a stdio checklist /
+                    ACP plan
   session-persistence/   persistence capability family
     session-persistence/         durable persistence seam + write coordinator
     session-persistence-jsonl/    JSONL-sidecar backend
@@ -172,6 +176,16 @@ pnpm run demo:acp   # run examples/acp-agent — the coding agent as an ACP
                     # drive it from Zed or another ACP client)
 ```
 
+### Run the CI gates locally BEFORE marking a PR ready
+
+CI is the backstop, not the first place a gate runs. Before you open a non-draft PR or move one from draft to ready, run the same gates CI runs, on your own tree, and confirm they pass — do not lean on CI (or a Codex pass) to discover a red gate you could have caught locally. The CI-equivalent local run is:
+
+```sh
+pnpm run typecheck && pnpm run lint && pnpm run test:coverage && pnpm run test:snapshot && pnpm run doc-sync && pnpm run hygiene && pnpm run build
+```
+
+**`pnpm run test:coverage`, NOT `pnpm run test`, is the gating test command.** `pnpm run test` runs `vitest run` with no coverage; CI's node job runs `test:coverage`, which enforces a **per-file 100%** threshold on `packages/*/*/src`. A suite that is green under `test` can still fail CI on an uncovered line — and that uncovered line is often *dead code* the 100% gate is correctly flagging for deletion (see [§ Defensive patterns](#defensive-patterns-hard-won) "Line coverage is not behavior coverage"), not a missing test to bolt on. `hygiene` (knip + publint + workspace constraints + NodeNext types) and `test:snapshot` (keyless ACP replay) are likewise CI gates that `test` alone does not cover. When you rely on a Codex convergence pass for sign-off, check WHICH commands it ran: a pass that ran `test` but not `test:coverage`/`hygiene`/`doc-sync` has not exercised those gates.
+
 ## Secrets / .env
 
 Real-API e2e tests (`pnpm run test:e2e`) read `DEEPSEEK_API_KEY` (and optionally `DEEPSEEK_BASE_URL`) from the environment, or from a gitignored `.env` at the repo root loaded via Node's native `process.loadEnvFile()`:

+ 2 - 1
docs/architecture.md

@@ -147,6 +147,7 @@ forever:
       session('assistant/message' {content, usage?})     log records what tool dispatch uses
       each tool-call (sequential, abort-checked between calls):
         session('tool/call'); ctx.tools.execute()     ⟵ waterfall tools/execute
+          tool execution may append tool-owned session events, e.g. `todo/write`
         session('tool/result')
       drain steering → session('steering/message'); emit agent/steering
       emit agent/step-end
@@ -197,7 +198,7 @@ Every MVP feature (including the TODO-marked ones), with the mechanism that impl
 | System prompt configurability | `ctx.systemPrompt.section()` with ordering |
 | AGENTS.md (root) | a section provider reading the file |
 | AGENTS.md (subdir, on-touch) + file-change notices | `agent.inject()` from a watcher / tool-result listener |
-| Built-in tools (Read/Write/Edit/Bash/…) | `ctx.tools.register()`; schemas flow into the assembly automatically. **Bash: implemented** — `dsh-bash` (seam) + `dsh-bash-local` (subprocesses) + `dsh-tool-bash` (`bash`/`bash_output`/`bash_kill`, incl. background tasks) |
+| Built-in tools (Read/Write/Edit/Bash/…) | `ctx.tools.register()`; schemas flow into the assembly automatically. **Bash: implemented** — `dsh-bash` (seam) + `dsh-bash-local` (subprocesses) + `dsh-tool-bash` (`bash`/`bash_output`/`bash_kill`, incl. background tasks). **`todo_write`: implemented** — `dsh-tool-todo` writes the whole task list to the session log (`todo/write`), rendered as a stdio checklist / ACP `plan` |
 | ToolSearch / progressive disclosure | wrap `agent/request`, filter `req.tools` |
 | Tool sandbox (landlock / sandbox-exec) | wrap `tools/execute`, or implement a sandboxing `BashExecutor` (the dsh-bash seam) |
 | Permission system / AskUserQuestion | wrap `tools/execute` (veto or ask); register an ask tool |

+ 1 - 1
docs/cookbook/adding-a-package.md

@@ -16,7 +16,7 @@ packages/<group>/<pkg>/
   README.md        # service API, events, extension points, design notes
 ```
 
-Choose an existing group when one matches the package's role (`core`, `llm`, `bash`, `session-persistence`, `ui`, `util`, or `support`). A new group is allowed, but it is a pure container: no `package.json`, no source files, and packages still sit exactly one level below it.
+Choose an existing group when one matches the package's role (`core`, `llm`, `bash`, `compact`, `subagent`, `todo`, `session-persistence`, `ui`, `util`, or `support`). A new group is allowed, but it is a pure container: no `package.json`, no source files, and packages still sit exactly one level below it.
 
 package.json invariants (enforced by `pnpm run constraints` / `scripts/check-workspace-constraints.ts`): `private: true`, `version: 0.0.1`, `type: module`, `main: "lib/index.js"`, `types: "lib/types/index.d.ts"`, `exports["."].types: "./lib/types/index.d.ts"`, `exports["."].default: "./lib/index.js"`, `cordis` in BOTH peerDependencies and devDependencies (same range). Mirror every dsh peer dependency in devDependencies. `schemastery` goes in `dependencies` (it is a runtime validator), matching agent-loop. The `files` list is precise: `lib/index.js`, `lib/types/**/*.d.ts`, `lib/types/**/*.d.ts.map`, and `src`; do not publish `lib/types` JS or JS-map intermediates or stale root declaration files. CLI app packages with a package `bin` include `lib/bin.js` immediately after `lib/index.js` in `files`.
 

+ 2 - 2
docs/cordis-catalog/events-and-services.md

@@ -11,7 +11,7 @@ The **harness tier** below (the `@deepseek-ai/dsh-*` packages) is the vocabulary
 
 ## Events
 
-Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../architecture.md#cordis-waterfall-semantics-important)), **parallel** (awaited fan-out, no veto), **serial** (awaited, in registration order, no veto). The harness declares 25 events across 6 scopes.
+Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../architecture.md#cordis-waterfall-semantics-important)), **parallel** (awaited fan-out, no veto), **serial** (awaited, in registration order, no veto).
 
 ### `agent/*`
 
@@ -315,7 +315,7 @@ Source: [`packages/core/tools/src/index.ts:43`](../../packages/core/tools/src/in
 
 ## Services
 
-The 10 `ctx.<key>` services the harness provides. An abstract seam (e.g. `ctx.bash`) is implemented by a separate package; the interface is what consumers code against.
+The `ctx.<key>` services the harness provides. An abstract seam (e.g. `ctx.bash`) is implemented by a separate package; the interface is what consumers code against.
 
 ### `ctx.agentLoop` — `AgentLoop`
 

+ 1 - 1
docs/core-data-structures/core.md

@@ -214,7 +214,7 @@ type SessionEvent<T extends SessionEventType = SessionEventType> = {
 }[T]
 ```
 
-The eleven event variants (`turn/start`, `turn/end`, `step/start`, `step/end`, `user/message`, `context/message`, `assistant/chunk`, `assistant/message`, `tool/call`, `tool/result`, `steering/message`), the `deriveMessages()` projection rules, the `TurnTrigger`/`TurnEndReason` reasons, and the turn-enclosure invariant are on **[session.md](session.md)**. How the log is made durable — the `SessionPersistence` seam, JSONL/SQLite backends, the `session/flush` checkpoint, crash recovery, and `SessionHeader` — is on **[persistence.md](persistence.md)**.
+The twelve event variants (`turn/start`, `turn/end`, `step/start`, `step/end`, `user/message`, `context/message`, `assistant/chunk`, `assistant/message`, `tool/call`, `tool/result`, `steering/message`, `todo/write`), the `deriveMessages()` projection rules, the `TurnTrigger`/`TurnEndReason` reasons, and the turn-enclosure invariant are on **[session.md](session.md)**. How the log is made durable — the `SessionPersistence` seam, JSONL/SQLite backends, the `session/flush` checkpoint, crash recovery, and `SessionHeader` — is on **[persistence.md](persistence.md)**.
 
 ## The agent handle
 

+ 25 - 0
docs/core-data-structures/session.md

@@ -35,6 +35,31 @@ interface SessionEventMap {
   'tool/result': { turn: number; step: number; callId: CallId; content: ContentBlock[]; isError: boolean; error?: { name: string; code: string } }
   /** Steering content injected between steps of a running turn. */
   'steering/message': { turn: number; content: ContentBlock[]; source: MessageSource }
+  /**
+   * The agent's whole todo list, carried as a full snapshot and replaced
+   * wholesale on each write — the current list is the most recent `todo/write`
+   * (last-write-wins on replay, no fold). Appended by an owning agent via
+   * `session.append('todo/write', { todos })`.
+   *
+   * NOT a {@link SurfaceEventType}: it produces no LLM message and never reaches
+   * `deriveMessages()`, so it carries no `surfaceOp` and stays off the surface —
+   * it is durable, replayable UI state, distinct from the conversation history.
+   * It is a `SessionEventMap` member riding the existing `session/event` emit,
+   * not a first-class Cordis `interface Events` notification, so it has no
+   * cordis-catalog row.
+   */
+  'todo/write': { todos: TodoItem[] }
+}
+```
+
+### `TodoItem` — one todo-list entry
+
+The unit of the `todo/write` event's whole-list snapshot. Deliberately minimal — a `content` line and a three-state `status` (no id, priority, or `activeForm`): the list is replaced wholesale on every write, so entries need no stable identity, and the status triple is exactly the ACP `PlanEntryStatus`, so a UI bridge can map a todo list onto an ACP `plan` 1:1 (synthesizing the priority ACP additionally requires). See the [todo_write RFC](../rfc/implemented/feature/2026-06-29-todo-write-tool.md).
+
+```ts type-equiv
+export interface TodoItem {
+  content: string
+  status: 'pending' | 'in_progress' | 'completed'
 }
 ```
 

+ 4 - 0
docs/module-graph.md

@@ -58,6 +58,9 @@ graph TD
   tool-bash --> bash
   tool-bash --> llm
   tool-bash --> tools
+  tool-todo --> agent
+  tool-todo --> session
+  tool-todo --> tools
   agent-core --> agent
   agent-core --> agent-loop
   agent-core --> invariants
@@ -120,6 +123,7 @@ graph TD
 | `agent-loop` | `agent`, `llm`, `session`, `session-persistence`, `system-prompt`, `tools` |
 | `subagent` | `agent`, `llm`, `tools` |
 | `tool-bash` | `agent`, `bash`, `llm`, `tools` |
+| `tool-todo` | `agent`, `session`, `tools` |
 | `agent-core` | `agent`, `agent-loop`, `invariants`, `llm`, `session`, `system-prompt`, `tool-bash`, `tools` |
 | `subagent-acp` | `agent`, `llm`, `subagent` |
 | `subagent-inprocess` | `agent`, `llm`, `session`, `subagent` |

+ 1 - 0
docs/rfc/README.md

@@ -85,6 +85,7 @@ Do NOT write one for a mechanical or local choice (a variable name, a one-file r
 | [Compaction as a capability seam (abstract contract + basic backend)](implemented/feature/2026-06-18-compaction-capability-seam.md) | 2026-06-18 |
 | [Subagent capability seam](implemented/feature/2026-06-21-subagent-capability-seam.md) | 2026-06-21 |
 | [ACP subagent backend (out-of-process delegation)](implemented/feature/2026-06-22-acp-subagent-backend.md) | 2026-06-22 |
+| [The `todo_write` tool — model task list as event-sourced session state](implemented/feature/2026-06-29-todo-write-tool.md) | 2026-06-29 |
 
 ### Simplification
 

+ 58 - 0
docs/rfc/implemented/feature/2026-06-29-todo-write-tool.md

@@ -0,0 +1,58 @@
+# RFC: The `todo_write` tool — model task list as event-sourced session state
+
+Status: implemented
+
+## Problem
+
+The harness gives the model bash and subagent tools but no way to record a structured task list. A todo list serves two co-equal purposes: it steers the model to plan multi-step work and keep the active task unambiguous (at most one active, exactly one while work remains), and it gives the human a live progress checklist. The ACP protocol has a native `plan` sessionUpdate that editors (Zed) already render, but the bridge never emitted one. Every reference coding agent surveyed (claude-code, opencode, codex, oh-my-pi, pi) ships some form of this; the harness had nothing.
+
+## Decision
+
+Add a model-facing `todo_write(todos: [{ content, status }])` tool whose whole-list state lives on the event-sourced session log as a new `todo/write` `SessionEventMap` variant. Both the stdio UI and the ACP bridge render off the existing `session/event` — the ACP bridge maps the list to a `plan` sessionUpdate.
+
+### Whole-list replace, three-state status
+
+The model sends the ENTIRE list every call; the new list replaces the old (last-write-wins on replay). This is the shape claude-code V1, opencode, and codex `update_plan` all use, and the shape the model is most trained on — no per-item ids, no delta protocol. `status` is exactly `pending | in_progress | completed`: the same triple as codex `update_plan` and, crucially, **identical to the ACP `PlanEntryStatus`**, so the bridge maps it 1:1 with no lossy translation.
+
+### State on the session log, not a service
+
+The list is appended as a `todo/write` event carrying the full `{ todos }` snapshot. The harness is event-sourced — the LLM history, tool calls, and turn structure all live on the log — so the todo list lives there too. This buys durability, replay, and `session/load` reconstruction for free: a reopened session re-derives the current list (the last `todo/write`) and the ACP bridge re-emits the `plan` on load, with no separate persistence backend, no in-memory service to rehydrate, and no extra wiring. An in-memory `ctx.todos` service would have had to reinvent all of that.
+
+### NOT a surface event
+
+`todo/write` is deliberately excluded from `SurfaceEventType`. The surface is the projection that produces the LLM message history (`deriveMessages()`); a todo write produces no conversation message. So it carries no `surfaceOp`, never joins the surface linked list, and never reaches `deriveMessages()` — it is durable, replayable *UI* state that travels alongside the conversation without being part of it. (The dev-mode invariants still require it to sit inside an open turn, which it always does: it is appended mid-step during a tool call.)
+
+### Priority synthesized only at the ACP boundary
+
+ACP's `PlanEntry` requires `content` + `priority` + `status`, but a `TodoItem` has no priority — the model never reasons about it. Rather than burden the schema with a field the model must always supply, the bridge synthesizes a constant `priority: 'medium'` on every entry when it builds the `plan`. Priority is an ACP wire requirement, not a harness concept, so it lives at exactly the boundary that needs it.
+
+### Dropped vs claude-code V1: `activeForm`, id, priority
+
+claude-code V1's item is `{ content, status, activeForm }`; later (V2) it grew ids, dependencies, and ownership — but only to support agent *swarms* (disk-backed, lock-guarded, per-item mutation). This tool keeps the item at the minimum: `{ content, status }`. No `activeForm` (the present-continuous label) — the UI shows `content`; no id — whole-list replace needs no stable identity; no priority — see above. Each dropped field is one less thing the model must produce on every call.
+
+### Single owner — no swarm machinery (YAGNI)
+
+The list belongs to the ONE agent session that called the tool (`exec.agent.session`); a non-agent caller is rejected. There is deliberately no shared/multi-owner scope, no capability seam (interface/impl/consumer), no scope resolver, and no delta protocol. The harness does have subagents, and a shared cross-agent list is conceivable — but building that now means designing for a form the product does not yet have. The whole-list-replace + single-owner shape is what claude-code V1, opencode, and codex all ship; if a shared list is ever needed, the on-log representation would change to per-item deltas (so concurrent writers can't clobber each other) and a scope resolver would choose the target log. That is a future RFC, not speculative scaffolding today.
+
+### Validation: the cheap middle
+
+The schema enforces type/required/enum. Beyond that, `execute` rejects empty or duplicate `content` and more than one `in_progress` task. claude-code leaves single-in-progress to the prompt; oh-my-pi enforces it in code. We take the middle: enforce the cheap invariants that make a plan *coherent* (no blank tasks, no dupes, at most one active), but leave ordering and the discipline of keeping the list current to the model via the tool description. A rejected write returns an `isError` result so the model self-corrects.
+
+## Why no cordis-catalog entry / no `@mode`
+
+`todo/write` is a member of `SessionEventMap`, not a first-class cordis `interface Events` event. The catalog generator (`scripts/gen-cordis-catalog.ts`) scans `interface Events` declarations; a `SessionEventMap` variant rides the existing `session/event` emit and produces no new catalog row. So it carries no `@mode` tag (which the generator requires only on `interface Events` members) — adding one would be meaningless.
+
+## Testing
+
+Four tiers, designed up front:
+- **Unit** — the session event (append/snapshot-clone/last-write-wins/not-on-surface); the tool (schema shape, arg validation via the real `ctx.tools.execute`, value validation, the event append + replacement, no-agent rejection, `presentCall`, HMR-safety); the ACP `todosToPlan` mapping; the stdio render arm.
+- **Real-Loader path** — the plugin run through `Loader.unwrapExports`, asserting the namespace export shape survives (it HAS `inject`, so a stray default would crash at load — postmortem/0001).
+- **Full-loop integration** — a scripted mock model calls `todo_write` through the real agent loop; the `todo/write` event lands and a second call replaces it.
+- **`session/load` replay** — a persisted `todo/write` re-emits the `plan` update when a fresh ACP bridge loads the session.
+- **With-key e2e + snapshot** — a real prompt induces a `todo_write`; the snapshot golden gains the `plan` notification and the log event.
+
+## Alternatives rejected
+
+- **In-memory `ctx.todos` service** — would reinvent durability, replay, and `session/load` reconstruction the log gives for free.
+- **Per-item delta protocol** — only needed for a shared multi-owner list, which is out of scope; whole-list replace is simpler and matches the references.
+- **Tool in `core/`** — `todo_write` is an extension tool registering on `ctx.tools`, not part of the spine; it lives in its own `packages/todo/` group like other tool families.

+ 1 - 1
examples/AGENTS.md

@@ -20,7 +20,7 @@ A keyless smoke that spawns the example from a temp cwd must set `TSX_TSCONFIG_P
 | Example | Keyless smoke | With-key smoke |
 |---|---|---|
 | `echo-agent` | `tests/echo.e2e.ts` — boots the real `cordis.yml`, drives the echo tool round-trip and the direct canned reply | **N/A — keyless by nature** (the `mock-echo` model has no real provider) |
-| `coding-agent` | `tests/keyless-smoke.e2e.ts` — boots the full real tree (dummy key, no prompt → no model call), asserts banner + clean exit | `tests/{full-loop,coding-task,resume,compaction}.e2e.ts` — real model + real bash, world-verified |
+| `coding-agent` | `tests/keyless-smoke.e2e.ts` — boots the full real tree (dummy key, no prompt → no model call), asserts banner + clean exit | `tests/{full-loop,coding-task,resume,compaction,todo-write}.e2e.ts` — real model + real bash + real todo_write, world-verified |
 | `acp-agent` | `pnpm run test:snapshot` — boots the real ACP subprocess and replays a recorded session keyless; `tests/acp.e2e.ts` also asserts stdout purity without a key | `tests/acp.e2e.ts` — real ACP prompt, verifies a file the agent wrote |
 
 See [the root AGENTS.md](../AGENTS.md) for repo-wide conventions and [docs/architecture.md](../docs/architecture.md) for the design.

+ 1 - 1
examples/README.md

@@ -15,7 +15,7 @@ Run with: `pnpm run demo:echo`. When prompted, type "echo <something>" to trigge
 
 ## coding-agent
 
-The real thing: DeepSeek V4 + the bash tool suite on the same `@deepseek-ai/dsh-stdio-agent` app. Where echo-agent proves the skeleton with mocks, this is a usable coding assistant.
+The real thing: DeepSeek V4 + the bash tool suite, `subagent` delegation, and the `todo_write` task tracker on the same `@deepseek-ai/dsh-stdio-agent` app. Where echo-agent proves the skeleton with mocks, this is a usable coding assistant.
 
 Run with: `pnpm run demo:coding` (needs `DEEPSEEK_API_KEY` in the environment or a gitignored repo-root `.env`). See [coding-agent/README.md](coding-agent/README.md) for details.
 

+ 14 - 1
examples/acp-agent/cordis.snapshot.yml

@@ -16,7 +16,9 @@
 - id: llm-replay
   name: '@deepseek-ai/dsh-llm-replay'
 
-# Local bash executor (the agent's only tool, via agent-core's tool-bash schema).
+# Local bash executor for agent-core's tool-bash schema.
+# FIXME(config-comments): keep this executor note from implying bash is the
+# whole tool set; subagent and todo_write are loaded below.
 - id: bash
   name: '@deepseek-ai/dsh-bash-local'
   config:
@@ -42,6 +44,12 @@
       a fresh child agent (it works in its own context and returns only its
       final result) — give it a complete, standalone instruction.
 
+      For multi-step work, use the todo_write tool to track a task list:
+      send the WHOLE list each call (it replaces the previous one), keep at
+      most one task in_progress (exactly one while work remains), and mark a
+      task completed as soon as it is done. Skip it for trivial single-step
+      tasks.
+
 # The subagent seam + both in-process backends + two model-facing tools —
 # identical to cordis.yml's wiring (only the LLM backend differs above): spawn
 # and fork are each reachable via a dsh-tool-subagent bound to it with a distinct
@@ -70,3 +78,8 @@
   config:
     provider: fork
     toolName: subagent_fork
+
+# The model-facing todo_write tool — identical to cordis.yml's wiring, so a
+# replayed todo_write tool call resolves to a real tool during snapshot replay.
+- id: tool-todo
+  name: '@deepseek-ai/dsh-tool-todo'

+ 14 - 1
examples/acp-agent/cordis.yml

@@ -23,7 +23,9 @@
       - deepseek-v4-flash
       - deepseek-v4-pro
 
-# Local bash executor (the agent's only tool, via agent-core's tool-bash schema).
+# Local bash executor for agent-core's tool-bash schema.
+# FIXME(config-comments): keep this executor note from implying bash is the
+# whole tool set; subagent and todo_write are loaded below.
 - id: bash
   name: '@deepseek-ai/dsh-bash-local'
   config:
@@ -51,6 +53,12 @@
       a fresh child agent (it works in its own context and returns only its
       final result) — give it a complete, standalone instruction.
 
+      For multi-step work, use the todo_write tool to track a task list:
+      send the WHOLE list each call (it replaces the previous one), keep at
+      most one task in_progress (exactly one while work remains), and mark a
+      task completed as soon as it is done. Skip it for trivial single-step
+      tasks.
+
 # The subagent seam + both in-process backends + two model-facing tools, as leaf
 # entries after the app (which provides ctx.agents/ctx.tools). spawn (a fresh
 # child) and fork (a child seeded with the parent's completed-turn prefix) are
@@ -81,3 +89,8 @@
   config:
     provider: fork
     toolName: subagent_fork
+
+# The model-facing todo_write tool: whole-list task tracking written to the
+# session log (todo/write), surfaced to the ACP client as a `plan` update.
+- id: tool-todo
+  name: '@deepseek-ai/dsh-tool-todo'

+ 1 - 0
examples/acp-agent/tests/acp.snapshot.ts

@@ -52,6 +52,7 @@ const SCENARIOS: Scenario[] = [
   { name: 'reject-extra-dirs', hasModelTurn: false, recorded: false },
   { name: 'text-turn', hasModelTurn: true, recorded: true },
   { name: 'tool-call-turn', hasModelTurn: true, recorded: true },
+  { name: 'todo-plan', hasModelTurn: true, recorded: true },
   { name: 'workspace-edit', hasModelTurn: true, recorded: true },
   { name: 'multi-turn', hasModelTurn: true, recorded: true },
   { name: 'error-finish', hasModelTurn: true, recorded: false },

+ 7 - 0
examples/acp-agent/tests/snapshots/todo-plan/input.json

@@ -0,0 +1,7 @@
+{
+  "steps": [
+    { "op": "initialize" },
+    { "op": "newSession" },
+    { "op": "prompt", "text": "Use the todo_write tool to record a plan with exactly three todos: \"read the code\" (in_progress), \"write the fix\" (pending), \"run the tests\" (pending). Send all three in one todo_write call. Then reply with the single word DONE and stop." }
+  ]
+}

+ 124 - 0
examples/acp-agent/tests/snapshots/todo-plan/session.jsonl

@@ -0,0 +1,124 @@
+{"type":"session","version":0,"id":"259ed557-03cf-4f50-9592-fc7fdbece7f3","createdAt":1782701599718,"cwd":"/tmp/acp-snap-cwd-4xZzZ9"}
+{"type":"turn/start","seq":0,"time":1782701599722,"data":{"turn":1,"trigger":{"kind":"message","source":{"kind":"user"}}}}
+{"type":"user/message","seq":1,"time":1782701599722,"data":{"content":[{"type":"text","text":"Use the todo_write tool to record a plan with exactly three todos: \"read the code\" (in_progress), \"write the fix\" (pending), \"run the tests\" (pending). Send all three in one todo_write call. Then reply with the single word DONE and stop."}],"source":{"kind":"user"}},"surfaceOp":"append"}
+{"type":"step/start","seq":2,"time":1782701599722,"data":{"turn":1,"step":1}}
+{"type":"assistant/chunk","seq":3,"time":1782701600164,"data":{"turn":1,"step":1,"chunk":{"type":"block-start","index":0,"blockType":"reasoning"}}}
+{"type":"assistant/chunk","seq":4,"time":1782701600164,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":"The"}}}
+{"type":"assistant/chunk","seq":5,"time":1782701600271,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" user"}}}
+{"type":"assistant/chunk","seq":6,"time":1782701600298,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" wants"}}}
+{"type":"assistant/chunk","seq":7,"time":1782701600299,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" me"}}}
+{"type":"assistant/chunk","seq":8,"time":1782701600299,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" to"}}}
+{"type":"assistant/chunk","seq":9,"time":1782701600299,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" use"}}}
+{"type":"assistant/chunk","seq":10,"time":1782701600299,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" todo"}}}
+{"type":"assistant/chunk","seq":11,"time":1782701600325,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":"_write"}}}
+{"type":"assistant/chunk","seq":12,"time":1782701600326,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" to"}}}
+{"type":"assistant/chunk","seq":13,"time":1782701600326,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" record"}}}
+{"type":"assistant/chunk","seq":14,"time":1782701600354,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" a"}}}
+{"type":"assistant/chunk","seq":15,"time":1782701600354,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" plan"}}}
+{"type":"assistant/chunk","seq":16,"time":1782701600355,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" with"}}}
+{"type":"assistant/chunk","seq":17,"time":1782701600355,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" exactly"}}}
+{"type":"assistant/chunk","seq":18,"time":1782701600355,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" three"}}}
+{"type":"assistant/chunk","seq":19,"time":1782701600383,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" todos"}}}
+{"type":"assistant/chunk","seq":20,"time":1782701600383,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":","}}}
+{"type":"assistant/chunk","seq":21,"time":1782701600409,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" then"}}}
+{"type":"assistant/chunk","seq":22,"time":1782701600438,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" reply"}}}
+{"type":"assistant/chunk","seq":23,"time":1782701600438,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" with"}}}
+{"type":"assistant/chunk","seq":24,"time":1782701600438,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":" D"}}}
+{"type":"assistant/chunk","seq":25,"time":1782701600465,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":"ONE"}}}
+{"type":"assistant/chunk","seq":26,"time":1782701600466,"data":{"turn":1,"step":1,"chunk":{"type":"reasoning-delta","index":0,"text":"."}}}
+{"type":"assistant/chunk","seq":27,"time":1782701600568,"data":{"turn":1,"step":1,"chunk":{"type":"block-start","index":1,"blockType":"tool-call"}}}
+{"type":"assistant/chunk","seq":28,"time":1782701600568,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":""}}}
+{"type":"assistant/chunk","seq":29,"time":1782701600568,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"{"}}}
+{"type":"assistant/chunk","seq":30,"time":1782701600568,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\""}}}
+{"type":"assistant/chunk","seq":31,"time":1782701600576,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"t"}}}
+{"type":"assistant/chunk","seq":32,"time":1782701600577,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"odos"}}}
+{"type":"assistant/chunk","seq":33,"time":1782701600577,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\""}}}
+{"type":"assistant/chunk","seq":34,"time":1782701600577,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":": "}}}
+{"type":"assistant/chunk","seq":35,"time":1782701600604,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"["}}}
+{"type":"assistant/chunk","seq":36,"time":1782701600605,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"{\""}}}
+{"type":"assistant/chunk","seq":37,"time":1782701600605,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"content"}}}
+{"type":"assistant/chunk","seq":38,"time":1782701600605,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\":"}}}
+{"type":"assistant/chunk","seq":39,"time":1782701600605,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":40,"time":1782701600631,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"read"}}}
+{"type":"assistant/chunk","seq":41,"time":1782701600631,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" the"}}}
+{"type":"assistant/chunk","seq":42,"time":1782701600632,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" code"}}}
+{"type":"assistant/chunk","seq":43,"time":1782701600632,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\","}}}
+{"type":"assistant/chunk","seq":44,"time":1782701600632,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":45,"time":1782701600632,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"status"}}}
+{"type":"assistant/chunk","seq":46,"time":1782701600659,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\":"}}}
+{"type":"assistant/chunk","seq":47,"time":1782701600660,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":48,"time":1782701600660,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"in"}}}
+{"type":"assistant/chunk","seq":49,"time":1782701600660,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"_pro"}}}
+{"type":"assistant/chunk","seq":50,"time":1782701600660,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"gress"}}}
+{"type":"assistant/chunk","seq":51,"time":1782701600660,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\"},"}}}
+{"type":"assistant/chunk","seq":52,"time":1782701600686,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" {\""}}}
+{"type":"assistant/chunk","seq":53,"time":1782701600687,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"content"}}}
+{"type":"assistant/chunk","seq":54,"time":1782701600687,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\":"}}}
+{"type":"assistant/chunk","seq":55,"time":1782701600687,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":56,"time":1782701600687,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"write"}}}
+{"type":"assistant/chunk","seq":57,"time":1782701600687,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" the"}}}
+{"type":"assistant/chunk","seq":58,"time":1782701600715,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" fix"}}}
+{"type":"assistant/chunk","seq":59,"time":1782701600715,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\","}}}
+{"type":"assistant/chunk","seq":60,"time":1782701600715,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":61,"time":1782701600715,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"status"}}}
+{"type":"assistant/chunk","seq":62,"time":1782701600716,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\":"}}}
+{"type":"assistant/chunk","seq":63,"time":1782701600716,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":64,"time":1782701600744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"pending"}}}
+{"type":"assistant/chunk","seq":65,"time":1782701600744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\"},"}}}
+{"type":"assistant/chunk","seq":66,"time":1782701600744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" {\""}}}
+{"type":"assistant/chunk","seq":67,"time":1782701600744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"content"}}}
+{"type":"assistant/chunk","seq":68,"time":1782701600744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\":"}}}
+{"type":"assistant/chunk","seq":69,"time":1782701600744,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":70,"time":1782701600770,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"run"}}}
+{"type":"assistant/chunk","seq":71,"time":1782701600770,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" the"}}}
+{"type":"assistant/chunk","seq":72,"time":1782701600770,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" tests"}}}
+{"type":"assistant/chunk","seq":73,"time":1782701600770,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\","}}}
+{"type":"assistant/chunk","seq":74,"time":1782701600770,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":75,"time":1782701600771,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"status"}}}
+{"type":"assistant/chunk","seq":76,"time":1782701600798,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\":"}}}
+{"type":"assistant/chunk","seq":77,"time":1782701600799,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":" \""}}}
+{"type":"assistant/chunk","seq":78,"time":1782701600799,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"pending"}}}
+{"type":"assistant/chunk","seq":79,"time":1782701600799,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"\""}}}
+{"type":"assistant/chunk","seq":80,"time":1782701600799,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"}]"}}}
+{"type":"assistant/chunk","seq":81,"time":1782701600826,"data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":1,"id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","argumentsDelta":"}"}}}
+{"type":"assistant/chunk","seq":82,"time":1782701600884,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"reasoning","text":"The user wants me to use todo_write to record a plan with exactly three todos, then reply with DONE."}}}}
+{"type":"assistant/chunk","seq":83,"time":1782701600884,"data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":1,"block":{"type":"tool-call","id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","arguments":"{\"todos\": [{\"content\": \"read the code\", \"status\": \"in_progress\"}, {\"content\": \"write the fix\", \"status\": \"pending\"}, {\"content\": \"run the tests\", \"status\": \"pending\"}]}"}}}}
+{"type":"assistant/chunk","seq":84,"time":1782701600884,"data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":1706,"outputTokens":113,"cacheReadTokens":0,"reasoningTokens":23}}}}
+{"type":"assistant/chunk","seq":85,"time":1782701600885,"data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"tool-calls"}}}}
+{"type":"assistant/message","seq":86,"time":1782701600886,"data":{"turn":1,"step":1,"content":[{"type":"reasoning","text":"The user wants me to use todo_write to record a plan with exactly three todos, then reply with DONE."},{"type":"tool-call","id":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","arguments":"{\"todos\": [{\"content\": \"read the code\", \"status\": \"in_progress\"}, {\"content\": \"write the fix\", \"status\": \"pending\"}, {\"content\": \"run the tests\", \"status\": \"pending\"}]}"}],"usage":{"inputTokens":1706,"outputTokens":113,"cacheReadTokens":0,"reasoningTokens":23}},"sourceEventSeqs":[3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85],"surfaceOp":"append"}
+{"type":"tool/call","seq":87,"time":1782701600887,"data":{"turn":1,"step":1,"callId":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","name":"todo_write","arguments":"{\"todos\": [{\"content\": \"read the code\", \"status\": \"in_progress\"}, {\"content\": \"write the fix\", \"status\": \"pending\"}, {\"content\": \"run the tests\", \"status\": \"pending\"}]}"}}
+{"type":"todo/write","seq":88,"time":1782701600887,"data":{"todos":[{"content":"read the code","status":"in_progress"},{"content":"write the fix","status":"pending"},{"content":"run the tests","status":"pending"}]}}
+{"type":"tool/result","seq":89,"time":1782701600887,"data":{"turn":1,"step":1,"callId":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","content":[{"type":"text","text":"Updated todo list: 2 pending, 1 in progress, 0 completed."}],"isError":false},"sourceEventSeqs":[87],"surfaceOp":"append"}
+{"type":"step/end","seq":90,"time":1782701600888,"data":{"turn":1,"step":1}}
+{"type":"step/start","seq":91,"time":1782701600888,"data":{"turn":1,"step":2}}
+{"type":"assistant/chunk","seq":92,"time":1782701601276,"data":{"turn":1,"step":2,"chunk":{"type":"block-start","index":0,"blockType":"reasoning"}}}
+{"type":"assistant/chunk","seq":93,"time":1782701601276,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":"The"}}}
+{"type":"assistant/chunk","seq":94,"time":1782701601382,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" todo"}}}
+{"type":"assistant/chunk","seq":95,"time":1782701601410,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" list"}}}
+{"type":"assistant/chunk","seq":96,"time":1782701601437,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" was"}}}
+{"type":"assistant/chunk","seq":97,"time":1782701601438,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" set"}}}
+{"type":"assistant/chunk","seq":98,"time":1782701601438,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" successfully"}}}
+{"type":"assistant/chunk","seq":99,"time":1782701601466,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":"."}}}
+{"type":"assistant/chunk","seq":100,"time":1782701601467,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" Now"}}}
+{"type":"assistant/chunk","seq":101,"time":1782701601467,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" I"}}}
+{"type":"assistant/chunk","seq":102,"time":1782701601467,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" just"}}}
+{"type":"assistant/chunk","seq":103,"time":1782701601494,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" need"}}}
+{"type":"assistant/chunk","seq":104,"time":1782701601494,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" to"}}}
+{"type":"assistant/chunk","seq":105,"time":1782701601495,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" reply"}}}
+{"type":"assistant/chunk","seq":106,"time":1782701601495,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" with"}}}
+{"type":"assistant/chunk","seq":107,"time":1782701601520,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" the"}}}
+{"type":"assistant/chunk","seq":108,"time":1782701601521,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" single"}}}
+{"type":"assistant/chunk","seq":109,"time":1782701601521,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" word"}}}
+{"type":"assistant/chunk","seq":110,"time":1782701601521,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":" D"}}}
+{"type":"assistant/chunk","seq":111,"time":1782701601550,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":"ONE"}}}
+{"type":"assistant/chunk","seq":112,"time":1782701601550,"data":{"turn":1,"step":2,"chunk":{"type":"reasoning-delta","index":0,"text":"."}}}
+{"type":"assistant/chunk","seq":113,"time":1782701601550,"data":{"turn":1,"step":2,"chunk":{"type":"block-start","index":1,"blockType":"text"}}}
+{"type":"assistant/chunk","seq":114,"time":1782701601551,"data":{"turn":1,"step":2,"chunk":{"type":"text-delta","index":1,"text":"D"}}}
+{"type":"assistant/chunk","seq":115,"time":1782701601551,"data":{"turn":1,"step":2,"chunk":{"type":"text-delta","index":1,"text":"ONE"}}}
+{"type":"assistant/chunk","seq":116,"time":1782701601551,"data":{"turn":1,"step":2,"chunk":{"type":"block-end","index":0,"block":{"type":"reasoning","text":"The todo list was set successfully. Now I just need to reply with the single word DONE."}}}}
+{"type":"assistant/chunk","seq":117,"time":1782701601551,"data":{"turn":1,"step":2,"chunk":{"type":"block-end","index":1,"block":{"type":"text","text":"DONE"}}}}
+{"type":"assistant/chunk","seq":118,"time":1782701601551,"data":{"turn":1,"step":2,"chunk":{"type":"usage","usage":{"inputTokens":174,"outputTokens":23,"cacheReadTokens":1664,"reasoningTokens":20}}}}
+{"type":"assistant/chunk","seq":119,"time":1782701601551,"data":{"turn":1,"step":2,"chunk":{"type":"finish","reason":{"kind":"stop"}}}}
+{"type":"assistant/message","seq":120,"time":1782701601551,"data":{"turn":1,"step":2,"content":[{"type":"reasoning","text":"The todo list was set successfully. Now I just need to reply with the single word DONE."},{"type":"text","text":"DONE"}],"usage":{"inputTokens":174,"outputTokens":23,"cacheReadTokens":1664,"reasoningTokens":20}},"sourceEventSeqs":[92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119],"surfaceOp":"append"}
+{"type":"step/end","seq":121,"time":1782701601552,"data":{"turn":1,"step":2}}
+{"type":"turn/end","seq":122,"time":1782701601552,"data":{"turn":1,"reason":{"kind":"completed"}}}

+ 51 - 0
examples/acp-agent/tests/snapshots/todo-plan/stdout.golden.jsonl

@@ -0,0 +1,51 @@
+{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentInfo":{"name":"deepseek-harness-acp","version":"0.0.1"},"agentCapabilities":{"loadSession":true,"promptCapabilities":{"image":false,"audio":false,"embeddedContext":false}},"authMethods":[]}}
+{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}"}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"The"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" user"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" wants"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" me"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" to"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" use"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" todo"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"_write"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" to"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" record"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" a"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" plan"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" with"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" exactly"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" three"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" todos"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":","}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" then"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" reply"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" with"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" D"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"ONE"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"."}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call","toolCallId":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","title":"Update todo list","kind":"other","status":"in_progress","rawInput":[{"content":"read the code","status":"in_progress"},{"content":"write the fix","status":"pending"},{"content":"run the tests","status":"pending"}]}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"plan","entries":[{"content":"read the code","priority":"medium","status":"in_progress"},{"content":"write the fix","priority":"medium","status":"pending"},{"content":"run the tests","priority":"medium","status":"pending"}]}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"tool_call_update","toolCallId":"call_00_OK2ZF1DYrsKHQQtxfQlJ0810","status":"completed","content":[{"type":"content","content":{"type":"text","text":"Updated todo list: 2 pending, 1 in progress, 0 completed."}}]}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"The"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" todo"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" list"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" was"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" set"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" successfully"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"."}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" Now"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" I"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" just"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" need"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" to"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" reply"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" with"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" the"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" single"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" word"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":" D"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"ONE"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_thought_chunk","content":{"type":"text","text":"."}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"D"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","content":{"type":"text","text":"ONE"}}}}
+{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn"}}

+ 1 - 1
examples/coding-agent/README.md

@@ -12,7 +12,7 @@ The first REAL agent wiring: DeepSeek V4 + the bash tool suite + stdio chat
 pnpm run demo:coding
 ```
 
-Type a coding task. The agent's only tools are `bash` (+ `bash_output` / `bash_kill` for background tasks): file reads, writes, searches, and test runs all happen through shell commands, each in a fresh `bash -c` (the system prompt tells the model to pass `workdir` instead of `cd`). Reasoning streams dimmed; tool calls/results render inline.
+Type a coding task. The agent works through `bash` (+ `bash_output` / `bash_kill` for background tasks): file reads, writes, searches, and test runs all happen through shell commands, each in a fresh `bash -c` (the system prompt tells the model to pass `workdir` instead of `cd`). It can also delegate with `subagent`/`subagent_fork` and track multi-step work with `todo_write` (a whole-list task tracker rendered as a checklist). Reasoning streams dimmed; tool calls/results render inline.
 
 ```
 > fix the failing test in /path/to/project

+ 15 - 2
examples/coding-agent/cordis.yml

@@ -28,7 +28,9 @@
       - deepseek-v4-flash
       - deepseek-v4-pro
 
-# Local bash executor (the model's only tool, via agent-core's tool-bash schema).
+# Local bash executor for agent-core's tool-bash schema.
+# FIXME(config-comments): keep this executor note from implying bash is the
+# whole tool set; subagent and todo_write are loaded below.
 - id: bash
   name: '@deepseek-ai/dsh-bash-local'
   config:
@@ -44,7 +46,7 @@
     # under ./.sessions); unset starts a fresh session each run.
     resumeSessionId: !!js process.env.RESUME_SESSION_ID
     persistenceRoot: './.sessions'
-    welcome: 'coding-agent ready. Give it a coding task (its tools are bash and subagent).'
+    welcome: 'coding-agent ready. Give it a coding task (its tools are bash, subagent, and todo_write).'
     systemPrompt: |
       You are coding-agent, a CLI coding assistant.
 
@@ -65,6 +67,12 @@
       failures before moving on. Verify your work by running the code or
       tests. Keep answers brief and factual.
 
+      For multi-step work, use the todo_write tool to track a task list:
+      send the WHOLE list each call (it replaces the previous one), keep at
+      most one task in_progress (exactly one while work remains), and mark a
+      task completed as soon as it is done. Skip it for trivial single-step
+      tasks.
+
 # Automatic context compaction: when the derived history approaches the model's
 # context window, summarize an older range into a checkpoint so a long-running
 # or tool-heavy session keeps fitting. A leaf entry (needs ctx.llm + the
@@ -106,3 +114,8 @@
   config:
     provider: fork
     toolName: subagent_fork
+
+# The model-facing todo_write tool: whole-list task tracking written to the
+# session log (todo/write), rendered as a stdio checklist / ACP plan.
+- id: tool-todo
+  name: '@deepseek-ai/dsh-tool-todo'

+ 13 - 4
examples/coding-agent/tests/harness.ts

@@ -8,6 +8,7 @@ import AgentRegistry from '@deepseek-ai/dsh-agent'
 import AgentLoop, { ReactLoopAgent } from '@deepseek-ai/dsh-agent-loop'
 import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local'
 import * as ToolBash from '@deepseek-ai/dsh-tool-bash'
+import * as ToolTodo from '@deepseek-ai/dsh-tool-todo'
 import * as LlmDeepSeek from '@deepseek-ai/dsh-llm-deepseek'
 import SessionPersistenceJsonl from '@deepseek-ai/dsh-session-persistence-jsonl'
 import { BasicCompactService } from '@deepseek-ai/dsh-compact-basic'
@@ -15,14 +16,21 @@ import type { BasicCompactConfig } from '@deepseek-ai/dsh-compact-basic'
 
 /**
  * Shared harness for the coding-agent e2e suites: the full plugin stack
- * with the real DeepSeek adapter and the real bash tool. Lives outside the
- * *.e2e.ts pattern so importing it never re-registers another file's tests.
+ * with the real DeepSeek adapter and the real bash + todo_write tools. Lives
+ * outside the *.e2e.ts pattern so importing it never re-registers another
+ * file's tests.
  */
 
-export const SYSTEM_PROMPT = 'You are a coding agent. Your only tool is bash; '
-  + 'do file operations with cat/grep/heredocs, check [exit code: N] markers, '
+export const SYSTEM_PROMPT = 'You are a coding agent. Use bash for file operations '
+  + 'with cat/grep/heredocs; check [exit code: N] markers, '
   + 'and report results briefly.'
 
+/** System prompt for the todo_write e2e: nudges the model to plan with the tool. */
+export const TODO_SYSTEM_PROMPT = 'You are a coding agent. For multi-step work, '
+  + 'use the todo_write tool to track a task list: send the WHOLE list each call, '
+  + 'keep at most one task in_progress (exactly one while work remains), and mark '
+  + 'a task completed as soon as it is done.'
+
 /** Options for {@link codingHarness}. */
 export interface CodingHarnessOptions {
   /** Durable JSONL persistence root (the resume suite needs it; others stay file-free). */
@@ -46,6 +54,7 @@ export async function codingHarness(workdir: string, options: CodingHarnessOptio
   await ctx.plugin(LlmDeepSeek, { models: ['deepseek-v4-flash'] })
   await ctx.plugin(LocalBashExecutor, { cwd: workdir, timeoutMs: 30_000 })
   await ctx.plugin(ToolBash)
+  await ctx.plugin(ToolTodo)
   // Compaction is opt-in: only the compaction e2e loads it, with a lowered
   // contextWindow/retainTokens so a short real session crosses the threshold.
   if (options.compact !== undefined) await ctx.plugin(BasicCompactService, options.compact)

+ 49 - 0
examples/coding-agent/tests/todo-write.e2e.ts

@@ -0,0 +1,49 @@
+import { afterEach, describe, expect, it } from 'vitest'
+import type { Context } from 'cordis'
+import { AgentId } from '@deepseek-ai/dsh-agent'
+import { codingHarness, TODO_SYSTEM_PROMPT, waitForIdle } from './harness.ts'
+
+/**
+ * A REAL model drives the REAL todo_write tool: verify the WORLD (the session
+ * log gains a todo/write event whose snapshot the model actually produced), not
+ * the agent's self-report. Key-gated (see vitest.e2e.config.ts).
+ */
+
+let ctx: Context | undefined
+
+afterEach(async () => {
+  await ctx?.fiber.dispose()
+  ctx = undefined
+})
+
+describe.skipIf(!process.env.DEEPSEEK_API_KEY)('todo_write: real model records a plan', () => {
+  it('appends a todo/write event with the model-produced task list', async () => {
+    ctx = await codingHarness(process.cwd())
+    const agent = ctx.agentLoop.create(AgentId('e2e-todo'), {
+      model: 'deepseek-v4-flash',
+      systemPrompt: TODO_SYSTEM_PROMPT,
+    })
+
+    agent.send([{ type: 'text', text:
+      'Use the todo_write tool to record a plan of exactly two steps: first '
+      + '"inspect the failing test" (in_progress), then "apply the fix" (pending). '
+      + 'Send both in one todo_write call, then reply with the single word DONE.' }])
+    await waitForIdle(ctx, agent)
+
+    const events = [...agent.session.events]
+
+    // The model actually called the tool.
+    const calls = events.filter(event => event.type === 'tool/call')
+    expect(calls.some(event => event.data.name === 'todo_write')).toBe(true)
+
+    // And the tool wrote a todo/write event to the log — verify the WORLD.
+    const todoEvents = events.filter(event => event.type === 'todo/write')
+    expect(todoEvents.length).toBeGreaterThan(0)
+
+    const todos = (todoEvents.at(-1)!).data.todos
+    expect(todos).toEqual([
+      { content: 'inspect the failing test', status: 'in_progress' },
+      { content: 'apply the fix', status: 'pending' },
+    ])
+  }, 120_000)
+})

+ 3 - 0
packages/README.md

@@ -13,6 +13,7 @@ Packages are grouped by modular role at `packages/<group>/<pkg>/`. The group dir
 | [`bash/`](bash/README.md) | Bash capability family: the executor seam, a local impl, and the model-facing tool | Product — stable surface |
 | [`compact/`](compact/README.md) | Compaction capability family: the abstract seam + a basic backend (tool deferred) | Product — stable surface |
 | [`subagent/`](subagent/README.md) | Subagent capability family: the provider-registry seam and the model-facing delegation tool | Product — stable surface |
+| [`todo/`](todo/README.md) | Todo/planning family: the model-facing `todo_write` tool (whole-list task tracking on the session log) | Product — stable surface |
 | [`session-persistence/`](session-persistence/README.md) | Persistence capability family: the seam + JSONL/SQLite backends | Product — stable surface |
 | [`ui/`](ui/README.md) | Editor/client integration surfaces (the ACP bridge) | Product — stable surface |
 | [`support/`](support/README.md) | Dev/test/example infrastructure (invariants, stdio UI, replay adapter) | Support — lower compatibility expectations |
@@ -47,6 +48,7 @@ dsh-subagent-spawn ← dsh-subagent, dsh-agent, dsh-session, dsh-llm  (in-proces
 dsh-subagent-fork ← dsh-subagent-spawn, dsh-agent, dsh-session      (in-process child seeded from parent log)
 dsh-subagent-acp  ← dsh-subagent, dsh-agent, dsh-llm, @agentclientprotocol/sdk  (out-of-process child over ACP)
 dsh-tool-subagent ← dsh-subagent, dsh-tools, dsh-agent (model-facing delegation tool)
+dsh-tool-todo     ← dsh-tools, dsh-agent, dsh-session  (model-facing todo_write tool; whole list on the session log)
 dsh-agent-core    ← timer, dsh-llm, dsh-session, dsh-system-prompt, dsh-tools, dsh-agent, dsh-invariants, dsh-tool-bash, dsh-agent-loop  (the providerless spine, as one bundle plugin)
 dsh-stdio-agent   ← dsh-agent-core, dsh-ui-stdio, dsh-session-persistence-jsonl, dsh-agent, dsh-session  (stdio chat APP + bin)
 dsh-acp-agent     ← dsh-agent-core, dsh-acp, dsh-session-persistence-jsonl     (ACP server APP + bin)
@@ -87,6 +89,7 @@ The rule: **extension** plugins depend on interfaces, never on the concrete loop
 | `subagent-acp/` | `subagent` | Out-of-process backend: a child agent in a spawned subprocess, driven over the Agent Client Protocol | (registers on `ctx.subagents`) |
 | `subagent-mock/` | `support` | Scripted `SubagentProvider` for testing the seam through the real load path | (registers on `ctx.subagents`) |
 | `tool-subagent/` | `subagent` | Model-facing `subagent` delegation tool over `ctx.subagents` | (registers on `ctx.tools`) |
+| `tool-todo/` | `todo` | Model-facing `todo_write` tool; writes the whole task list to the session log (`todo/write`) | (registers on `ctx.tools`) |
 | `brand/` | `util` | Type-only `Branded<B>` nominal-typing primitive (no runtime code, no harness deps) | (none — type-only) |
 
 Each package has its own `README.md` with purpose, service API, events, extension points, and deliberate non-goals (TODOs).

+ 32 - 0
packages/core/session/src/types.ts

@@ -149,6 +149,24 @@ export interface TurnEndReasonMap {
 
 export type TurnEndReason = TurnEndReasonMap[keyof TurnEndReasonMap]
 
+/**
+ * One entry in an agent's todo list — the unit of the `todo/write`
+ * {@link SessionEventMap} event's whole-list snapshot.
+ *
+ * Deliberately minimal: a human-readable `content` line and a three-state
+ * `status`. No id, priority, or `activeForm` — the list is replaced wholesale
+ * on every write (last-write-wins), so entries need no stable identity, and the
+ * status triple is exactly the ACP `PlanEntryStatus`, so a UI bridge can map a
+ * todo list onto an ACP `plan` 1:1 (synthesizing the priority ACP additionally
+ * requires).
+ */
+export interface TodoItem {
+  /** What this task is — a short imperative line shown in the UI. */
+  content: string
+  /** Lifecycle state. `in_progress` marks the single task being worked now. */
+  status: 'pending' | 'in_progress' | 'completed'
+}
+
 /**
  * The session event vocabulary — the append-only source of truth for an
  * agent's whole interaction history. The LLM message history is *derived*
@@ -195,6 +213,20 @@ export interface SessionEventMap {
   'tool/result': { turn: number; step: number; callId: CallId; content: ContentBlock[]; isError: boolean; error?: { name: string; code: string } }
   /** Steering content injected between steps of a running turn. */
   'steering/message': { turn: number; content: ContentBlock[]; source: MessageSource }
+  /**
+   * The agent's whole todo list, carried as a full snapshot and replaced
+   * wholesale on each write — the current list is the most recent `todo/write`
+   * (last-write-wins on replay, no fold). Appended by an owning agent via
+   * `session.append('todo/write', { todos })`.
+   *
+   * NOT a {@link SurfaceEventType}: it produces no LLM message and never reaches
+   * `deriveMessages()`, so it carries no `surfaceOp` and stays off the surface —
+   * it is durable, replayable UI state, distinct from the conversation history.
+   * It is a `SessionEventMap` member riding the existing `session/event` emit,
+   * not a first-class Cordis `interface Events` notification, so it has no
+   * cordis-catalog row.
+   */
+  'todo/write': { todos: TodoItem[] }
 }
 
 export type SessionEventType = keyof SessionEventMap

+ 61 - 1
packages/core/session/tests/session.spec.ts

@@ -2,7 +2,7 @@ import { describe, expect, it } from 'vitest'
 import { Context } from 'cordis'
 import { CallId } from '@deepseek-ai/dsh-llm'
 import SessionStore, { SESSION_FORMAT_VERSION, Session, SessionEvent, SessionId } from '@deepseek-ai/dsh-session'
-import type { SessionEventType } from '@deepseek-ai/dsh-session'
+import type { SessionEventType, TodoItem } from '@deepseek-ai/dsh-session'
 
 describe('Session', () => {
   it('derives message history from the event log', () => {
@@ -365,3 +365,63 @@ describe('SessionStore', () => {
     expect(events).toHaveLength(1)
   })
 })
+
+describe('todo/write event', () => {
+  it('appends the whole-list snapshot and isolates the log from later mutation', () => {
+    const session = new Session(SessionId('t1'))
+    const todos: TodoItem[] = [
+      { content: 'plan the work', status: 'in_progress' },
+      { content: 'write the code', status: 'pending' },
+    ]
+    session.append('todo/write', { todos })
+
+    const event = session.events.findLast(e => e.type === 'todo/write')!
+    expect(event.type).toBe('todo/write')
+    expect(event.data.todos).toEqual(todos)
+
+    // The append snapshots its input: mutating the caller's array afterward must
+    // not change what the log holds (the durable-source-of-truth contract).
+    todos.push({ content: 'sneak in', status: 'pending' })
+    todos[0]!.status = 'completed'
+    expect(event.data.todos).toEqual([
+      { content: 'plan the work', status: 'in_progress' },
+      { content: 'write the code', status: 'pending' },
+    ])
+  })
+
+  it('is last-write-wins: the current list is the most recent todo/write', () => {
+    const session = new Session(SessionId('t2'))
+    session.append('todo/write', { todos: [{ content: 'first', status: 'pending' }] })
+    session.append('todo/write', { todos: [
+      { content: 'first', status: 'completed' },
+      { content: 'second', status: 'in_progress' },
+    ] })
+
+    const current = session.events.findLast(e => e.type === 'todo/write')!.data.todos
+    expect(current).toEqual([
+      { content: 'first', status: 'completed' },
+      { content: 'second', status: 'in_progress' },
+    ])
+  })
+
+  it('is NOT a surface event: it produces no derived message and joins no surface node', () => {
+    const session = new Session(SessionId('t3'))
+    session.append('user/message', { content: [{ type: 'text', text: 'q' }], source: { kind: 'user' } }, { surfaceOp: 'append' })
+    const before = session.deriveMessages().length
+    session.append('todo/write', { todos: [{ content: 'a task', status: 'pending' }] })
+    // The todo event must not add a message to the derived history…
+    expect(session.deriveMessages()).toHaveLength(before)
+    // …and must not appear on the surface linked list.
+    expect(session.surface.nodes.some(node => node.seq === session.seq - 1)).toBe(false)
+  })
+
+  it('round-trips through a seeded replay identically (durable, no surfaceOp needed)', () => {
+    const original = new Session(SessionId('t4'))
+    original.append('todo/write', { todos: [{ content: 'only', status: 'completed' }] })
+    // Seeding a non-surface event with no surfaceOp must not throw.
+    const replayed = new Session(SessionId('t4-replay'), [...original.events])
+    expect(replayed.events.findLast(e => e.type === 'todo/write')!.data.todos)
+      .toEqual([{ content: 'only', status: 'completed' }])
+    expect(replayed.seq).toBe(original.seq)
+  })
+})

+ 7 - 0
packages/support/ui-stdio/src/index.ts

@@ -110,6 +110,13 @@ export function createStdioChat(ctx: Context, config: Config, runtime: StdioRunt
       const { content } = event.data
       const text = content.filter(block => block.type === 'text').map(block => block.text).join('')
       output.write(`\n  [tool result] ${text}\n  `)
+    } else if (event.type === 'todo/write') {
+      if (inReasoning) output.write('\x1B[0m')
+      inReasoning = false
+      const glyph = (status: string): string =>
+        status === 'completed' ? '[x]' : status === 'in_progress' ? '[~]' : '[ ]'
+      const lines = event.data.todos.map(todo => `    ${glyph(todo.status)} ${todo.content}`).join('\n')
+      output.write(`\n  [todos]\n${lines}\n  `)
     }
   })
 

+ 29 - 0
packages/support/ui-stdio/tests/ui-stdio.spec.ts

@@ -151,6 +151,35 @@ describe('createStdioChat rendering', () => {
     expect(out.text()).toContain('[tool result] file.txt')
   })
 
+  it('renders a todo/write session event as a glyphed checklist', async () => {
+    const { ctx, out } = await setup()
+    const session = {} as Session
+    ctx.emit('session/event', session, {
+      type: 'todo/write', seq: 1, time: 0,
+      data: { todos: [
+        { content: 'read the code', status: 'completed' },
+        { content: 'write the fix', status: 'in_progress' },
+        { content: 'run the tests', status: 'pending' },
+      ] },
+    } as SessionEvent)
+    const text = out.text()
+    expect(text).toContain('[todos]')
+    expect(text).toContain('[x] read the code')
+    expect(text).toContain('[~] write the fix')
+    expect(text).toContain('[ ] run the tests')
+  })
+
+  it('resets dim styling when a todo/write interrupts reasoning', async () => {
+    const { ctx, out } = await setup()
+    const agent = makeAgent('main')
+    ctx.emit('agent/stream-chunk', agent, 1, 0, { type: 'reasoning-delta', index: 0, text: 'r' })
+    ctx.emit('session/event', {} as Session, {
+      type: 'todo/write', seq: 1, time: 0,
+      data: { todos: [{ content: 'a task', status: 'pending' }] },
+    } as SessionEvent)
+    expect(out.text()).toContain('\x1B[2mr\x1B[0m')
+  })
+
   it('resets dim styling when a tool/call interrupts reasoning', async () => {
     const { ctx, out } = await setup()
     const agent = makeAgent('main')

+ 9 - 0
packages/todo/README.md

@@ -0,0 +1,9 @@
+# todo/ — todo / planning capability family
+
+The model-facing todo tool. A single **product** package — there is no interface/implementation seam here, because the list is single-owner session state (one agent session owns its own list), not a swappable capability.
+
+| Package | Role | ctx key |
+|---|---|---|
+| `tool-todo/` | Model-facing `todo_write` tool; writes the whole list to the session log (`todo/write`) | (registers on `ctx.tools`) |
+
+The list lives on the event-sourced session log (`SessionEventMap['todo/write']`, owned by [`dsh-session`](../core/session)); this package is the thin consumer that appends the snapshot. UIs render off `session/event`: the [stdio UI](../support/ui-stdio) prints the list, the [ACP bridge](../ui/acp) maps it to a `plan` sessionUpdate.

+ 25 - 0
packages/todo/tool-todo/README.md

@@ -0,0 +1,25 @@
+# @deepseek-ai/dsh-tool-todo
+
+The model-facing `todo_write` tool: the agent's whole task list, replaced wholesale on each call.
+
+## What it does
+
+Registers one tool, `todo_write(todos: [{ content, status }])`, on `ctx.tools`. The model sends the ENTIRE list every call — there are no partial updates or per-item edits. Each call appends a `todo/write` event (the full list snapshot) to the calling agent's session log via `agent.session.append('todo/write', { todos })`; the current list is the most recent such event (last-write-wins on replay).
+
+`status` is one of `pending`, `in_progress`, `completed` — exactly the ACP `PlanEntryStatus` triple.
+
+## Single owner
+
+The list belongs to the ONE agent session that called the tool. There is no subagent/shared/swarm scope: a non-agent caller (no `exec.agent`) has nowhere to write the list and is rejected. This is a deliberate scope limit — see the RFC.
+
+## Validation
+
+Beyond the schema's type/required/enum checks, `execute` rejects an empty or duplicate `content` and more than one `in_progress` task (a coherent plan has at most one task active). Ordering and the discipline of keeping the list current are left to the model via the tool description.
+
+## Rendering
+
+The tool writes only the session event; it does not render. UIs subscribe to `session/event` and render the `todo/write` data themselves: the [stdio UI](../../support/ui-stdio) prints a glyphed checklist, and the [ACP bridge](../../ui/acp) maps the list to a `plan` sessionUpdate (synthesizing the `priority` ACP requires).
+
+## Export shape
+
+A function/namespace plugin: it exports `name` / `inject` / `apply` and NO default. A stray `export default` would collapse the module via the Loader's `unwrapExports` and drop `inject` (see [docs/postmortem/0001](../../../docs/postmortem/0001-acp-default-export-drops-inject.md)).

+ 39 - 0
packages/todo/tool-todo/package.json

@@ -0,0 +1,39 @@
+{
+  "name": "@deepseek-ai/dsh-tool-todo",
+  "description": "Model-facing todo_write tool over the DeepSeek Harness event-sourced session log",
+  "version": "0.0.1",
+  "private": true,
+  "type": "module",
+  "main": "lib/index.js",
+  "types": "lib/types/index.d.ts",
+  "exports": {
+    ".": {
+      "types": "./lib/types/index.d.ts",
+      "default": "./lib/index.js"
+    },
+    "./src/*": "./src/*",
+    "./package.json": "./package.json"
+  },
+  "files": [
+    "lib/index.js",
+    "lib/types/**/*.d.ts",
+    "lib/types/**/*.d.ts.map",
+    "src"
+  ],
+  "license": "BSD-3-Clause",
+  "peerDependencies": {
+    "@deepseek-ai/dsh-agent": "^0.0.1",
+    "@deepseek-ai/dsh-session": "^0.0.1",
+    "@deepseek-ai/dsh-tools": "^0.0.1",
+    "cordis": "^4.0.0-rc.6"
+  },
+  "devDependencies": {
+    "@deepseek-ai/dsh-agent": "workspace:^",
+    "@deepseek-ai/dsh-agent-loop": "workspace:^",
+    "@deepseek-ai/dsh-llm": "workspace:^",
+    "@deepseek-ai/dsh-session": "workspace:^",
+    "@deepseek-ai/dsh-system-prompt": "workspace:^",
+    "@deepseek-ai/dsh-tools": "workspace:^",
+    "cordis": "^4.0.0-rc.6"
+  }
+}

+ 123 - 0
packages/todo/tool-todo/src/index.ts

@@ -0,0 +1,123 @@
+/**
+ * The model-facing `todo_write` tool: the agent's whole task list, replaced
+ * wholesale on each call. Every call appends a `todo/write` event (the full
+ * list snapshot) to the calling agent's session log via
+ * `exec.agent.session.append('todo/write', { todos })`; the current list is the
+ * most recent such event (last-write-wins on replay). UIs render off
+ * `session/event`: the stdio UI prints the checklist, the ACP bridge maps it to
+ * a `plan` sessionUpdate.
+ *
+ * Single owner: the list belongs to the ONE agent session that called the tool.
+ * There is no subagent/shared/swarm scope — a non-agent caller (no
+ * `exec.agent`) has nowhere to write the list and is rejected.
+ *
+ * Plugin export shape: named exports, NO default. The cordis Loader's
+ * `unwrapExports` does `exports.default ?? exports`, so a stray default would
+ * collapse the module to the bare `apply` and drop `inject`, crashing at load
+ * (see docs/postmortem/0001).
+ *
+ * @module @deepseek-ai/dsh-tool-todo
+ */
+
+import type { Context } from 'cordis'
+import type { ContentBlock } from '@deepseek-ai/dsh-llm'
+import { defineTool } from '@deepseek-ai/dsh-tools'
+import type { TodoItem } from '@deepseek-ai/dsh-session'
+
+export const name = 'tool-todo'
+export const inject = ['tools']
+
+/** The valid {@link TodoItem} statuses, as a runtime set for input narrowing. */
+const STATUSES = ['pending', 'in_progress', 'completed'] as const
+
+const DESCRIPTION =
+  'Record and update a structured task list for the current work. Send the ENTIRE '
+  + 'list every call — it REPLACES the previous list (there are no partial updates, '
+  + 'no per-item edits). Use it to plan multi-step work and show progress: add one '
+  + 'todo per concrete step before you start. Keep AT MOST ONE todo `in_progress` '
+  + 'at a time; while work remains, exactly one active task should be '
+  + '`in_progress`. Mark a todo `completed` the moment it is done (do not batch '
+  + 'completions), and allow no `in_progress` item only once all work is complete. '
+  + 'Skip the list for trivial single-step tasks. Statuses: `pending` '
+  + '(not started), `in_progress` (being worked on now), `completed` (finished).'
+
+/**
+ * Validate the value constraints the SchemaSpec can't express and build the
+ * canonical {@link TodoItem}[].
+ *
+ * `defineTool` already validates type/required/enum before `execute` runs (a
+ * bad `status` is rejected by the registry's `validateArgs`, never reaching
+ * here), so `status` is guaranteed to be one of the three enum literals. But
+ * `InferArgs` maps an `enum` string prop to plain `string`, so the compiler sees
+ * `args.todos` as `{ content: string; status: string }[]`; the
+ * `status as TodoItem['status']` narrowing records that registry guarantee
+ * rather than re-checking it (an unreachable re-check would be dead code — see
+ * AGENTS.md "don't validate scenarios that can't happen"). What remains is the
+ * value rules the DSL has no vocabulary for: non-empty unique content (stored
+ * trimmed, so the persisted value matches the dedupe/length key), and at most
+ * one `in_progress` task.
+ */
+function toTodoList(raw: { content: string; status: string }[]): TodoItem[] {
+  const todos: TodoItem[] = []
+  const seen = new Set<string>()
+  let inProgress = 0
+  for (const item of raw) {
+    const content = item.content.trim()
+    if (content.length === 0) {
+      throw new Error('invalid todo: `content` must be a non-empty string')
+    }
+    if (seen.has(content)) {
+      throw new Error(`invalid todos: duplicate content ${JSON.stringify(content)}`)
+    }
+    seen.add(content)
+    const status = item.status as TodoItem['status']
+    if (status === 'in_progress') inProgress++
+    todos.push({ content, status })
+  }
+  if (inProgress > 1) {
+    throw new Error(`invalid todos: at most one task may be in_progress, got ${inProgress}`)
+  }
+  return todos
+}
+
+/** Register the `todo_write` tool on `ctx.tools`. */
+export function apply(ctx: Context): void {
+  ctx.tools.register(defineTool({
+    name: 'todo_write',
+    description: DESCRIPTION,
+    parameters: {
+      todos: {
+        type: 'array',
+        required: true,
+        description: 'The COMPLETE task list, replacing any previous list.',
+        items: {
+          type: 'object',
+          properties: {
+            content: { type: 'string', required: true, description: 'What the task is — a short imperative line.' },
+            status: {
+              type: 'string',
+              required: true,
+              enum: [...STATUSES],
+              description: 'pending (not started) | in_progress (now) | completed (done).',
+            },
+          },
+        },
+      },
+    },
+    execute(args, exec): Promise<ContentBlock[]> {
+      const todos = toTodoList(args.todos)
+      if (!exec.agent) {
+        // The list is per-agent-session state; a non-agent caller (no owning
+        // session) has nowhere to write it. Reject rather than silently no-op.
+        throw new Error('todo_write requires an owning agent session')
+      }
+      exec.agent.session.append('todo/write', { todos })
+      const count = (status: TodoItem['status']): number => todos.filter(t => t.status === status).length
+      return Promise.resolve([{
+        type: 'text',
+        text: `Updated todo list: ${count('pending')} pending, ${count('in_progress')} in progress, ${count('completed')} completed.`,
+      }])
+    },
+    presentCall: args => ({ title: 'Update todo list', kind: 'other', rawInput: args.todos }),
+  }))
+}

+ 107 - 0
packages/todo/tool-todo/tests/integration.spec.ts

@@ -0,0 +1,107 @@
+import { describe, expect, it } from 'vitest'
+import { Context } from 'cordis'
+import LlmService from '@deepseek-ai/dsh-llm'
+import SessionStore from '@deepseek-ai/dsh-session'
+import type { SessionEvent } from '@deepseek-ai/dsh-session'
+import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
+import ToolRegistry from '@deepseek-ai/dsh-tools'
+import AgentRegistry, { AgentId } from '@deepseek-ai/dsh-agent'
+import AgentLoop, { ReactLoopAgent } from '@deepseek-ai/dsh-agent-loop'
+import * as ToolTodo from '@deepseek-ai/dsh-tool-todo'
+import { MockAdapter, textResponse, toolCallResponse } from '../../../core/agent-loop/tests/mock-adapter.ts'
+
+/**
+ * Full-loop integration: a scripted mock model drives the REAL todo_write tool
+ * through the agent loop, exercising the same seams a live model would — the
+ * tool/call + tool/result session events AND the todo/write event the tool
+ * appends. Only the model is mocked; the tool and the session log are real.
+ */
+async function harness(adapter: MockAdapter): Promise<Context> {
+  const ctx = new Context()
+  await ctx.plugin(LlmService)
+  await ctx.plugin(SessionStore)
+  await ctx.plugin(SystemPrompt)
+  await ctx.plugin(ToolRegistry)
+  await ctx.plugin(AgentRegistry)
+  await ctx.plugin(AgentLoop, { agents: [] })
+  await ctx.plugin(ToolTodo)
+  ctx.llm.registerAdapter(['mock'], adapter)
+  return ctx
+}
+
+function waitForIdle(ctx: Context, agent: ReactLoopAgent): Promise<void> {
+  return new Promise((resolve) => {
+    const dispose = ctx.on('agent/status', (subject, status) => {
+      if (subject === agent && status === 'idle') {
+        dispose()
+        resolve()
+      }
+    })
+  })
+}
+
+function findEvent<T extends SessionEvent['type']>(
+  log: readonly SessionEvent[],
+  type: T,
+  position: 'first' | 'last' = 'first',
+): Extract<SessionEvent, { type: T }> {
+  const found = position === 'first'
+    ? log.find(event => event.type === type)
+    : log.findLast(event => event.type === type)
+  if (!found) throw new Error(`no ${type} event in the session log`)
+  return found as Extract<SessionEvent, { type: T }>
+}
+
+describe('todo_write tool through the agent loop', () => {
+  it('model calls todo_write: a tool/call, a non-error tool/result, and a todo/write snapshot land', async () => {
+    const adapter = new MockAdapter([
+      toolCallResponse('call-1', 'todo_write', {
+        todos: [
+          { content: 'read the code', status: 'in_progress' },
+          { content: 'write the fix', status: 'pending' },
+        ],
+      }, 'Planning the work.'),
+      textResponse('Plan recorded.'),
+    ])
+    const ctx = await harness(adapter)
+    const agent = ctx.agentLoop.create(AgentId('it-todo'), { model: 'mock' })
+
+    agent.send([{ type: 'text', text: 'plan a two-step task' }])
+    await waitForIdle(ctx, agent)
+
+    const log = agent.session.events
+    expect(findEvent(log, 'tool/call').data.name).toBe('todo_write')
+    expect(findEvent(log, 'tool/result').data.isError).toBe(false)
+
+    const todoEvent = findEvent(log, 'todo/write')
+    expect(todoEvent.data.todos).toEqual([
+      { content: 'read the code', status: 'in_progress' },
+      { content: 'write the fix', status: 'pending' },
+    ])
+  })
+
+  it('a second todo_write replaces the list (last-write-wins on the log)', async () => {
+    const adapter = new MockAdapter([
+      toolCallResponse('call-1', 'todo_write', { todos: [{ content: 'step one', status: 'in_progress' }] }),
+      toolCallResponse('call-2', 'todo_write', {
+        todos: [
+          { content: 'step one', status: 'completed' },
+          { content: 'step two', status: 'in_progress' },
+        ],
+      }),
+      textResponse('Done planning.'),
+    ])
+    const ctx = await harness(adapter)
+    const agent = ctx.agentLoop.create(AgentId('it-todo-2'), { model: 'mock' })
+
+    agent.send([{ type: 'text', text: 'plan then update' }])
+    await waitForIdle(ctx, agent)
+
+    const todoEvents = agent.session.events.filter(e => e.type === 'todo/write')
+    expect(todoEvents).toHaveLength(2)
+    expect(findEvent(agent.session.events, 'todo/write', 'last').data.todos).toEqual([
+      { content: 'step one', status: 'completed' },
+      { content: 'step two', status: 'in_progress' },
+    ])
+  })
+})

+ 167 - 0
packages/todo/tool-todo/tests/tool-todo.spec.ts

@@ -0,0 +1,167 @@
+import { describe, expect, it } from 'vitest'
+import { Context } from 'cordis'
+import Loader from '@cordisjs/plugin-loader'
+import { CallId } from '@deepseek-ai/dsh-llm'
+import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
+import ToolRegistry from '@deepseek-ai/dsh-tools'
+import { Session, SessionId } from '@deepseek-ai/dsh-session'
+import type { TodoItem } from '@deepseek-ai/dsh-session'
+import { AgentId, type Agent } from '@deepseek-ai/dsh-agent'
+import * as tool from '../src/index.ts'
+
+/**
+ * Drives the REAL plugin body: mounts `dsh-tool-todo` on a real `ToolRegistry`
+ * and invokes the registered `todo_write` tool through `ctx.tools.execute`,
+ * with a fake parent Agent carrying a real `Session` — so the append the tool
+ * makes is observable on a genuine session log (only the agent wrapper is a
+ * stand-in; the session and the tool are the shipping code).
+ */
+
+/** A parent Agent backed by a real Session — the tool reads `agent.session`. */
+function agentWithSession(id = 'parent-1'): Agent & { session: Session } {
+  const session = new Session(SessionId(id))
+  return { id: AgentId(id), session } as unknown as Agent & { session: Session }
+}
+
+async function setup(): Promise<Context> {
+  const ctx = new Context()
+  await ctx.plugin(SystemPrompt)
+  await ctx.plugin(ToolRegistry)
+  await ctx.plugin(tool)
+  return ctx
+}
+
+let callCounter = 0
+function callTodo(ctx: Context, args: unknown, over: { agent?: Agent | undefined } = {}) {
+  const agent = 'agent' in over ? over.agent : agentWithSession()
+  return ctx.tools.execute({
+    callId: CallId(`call-${++callCounter}`),
+    name: 'todo_write',
+    arguments: args,
+    ...agent ? { agent } : {},
+  })
+}
+
+function text(result: { content: { type: string; text?: string }[] }): string {
+  return result.content.filter(b => b.type === 'text').map(b => b.text).join('')
+}
+
+describe('dsh-tool-todo', () => {
+  it('registers a `todo_write` tool whose schema is an array of {content,status}', async () => {
+    const ctx = await setup()
+    const schema = ctx.tools.schemas().find(s => s.name === 'todo_write')
+    expect(schema).toBeDefined()
+    const props = (schema!.parameters as { properties?: Record<string, unknown> }).properties ?? {}
+    expect(Object.keys(props)).toEqual(['todos'])
+    const todos = props.todos as { type: string; items?: { properties?: Record<string, { type: string; enum?: string[] }> } }
+    expect(todos.type).toBe('array')
+    const itemProps = todos.items?.properties ?? {}
+    expect(Object.keys(itemProps).sort()).toEqual(['content', 'status'])
+    expect(itemProps.status?.enum).toEqual(['pending', 'in_progress', 'completed'])
+  })
+
+  it('appends a todo/write event carrying the whole list to the calling session', async () => {
+    const ctx = await setup()
+    const agent = agentWithSession('writer')
+    const todos: TodoItem[] = [
+      { content: 'plan', status: 'in_progress' },
+      { content: 'build', status: 'pending' },
+    ]
+    const result = await callTodo(ctx, { todos }, { agent })
+    expect(result.isError).toBe(false)
+    expect(text(result)).toContain('1 pending, 1 in progress, 0 completed')
+
+    const event = agent.session.events.findLast(e => e.type === 'todo/write')!
+    expect(event.data.todos).toEqual(todos)
+  })
+
+  it('stores the trimmed content (the dedupe/length key), not the raw input', async () => {
+    const ctx = await setup()
+    const agent = agentWithSession('trim')
+    const result = await callTodo(ctx, { todos: [{ content: '  plan the work  ', status: 'pending' }] }, { agent })
+    expect(result.isError).toBe(false)
+
+    const event = agent.session.events.findLast(e => e.type === 'todo/write')!
+    expect(event.data.todos).toEqual([{ content: 'plan the work', status: 'pending' }])
+  })
+
+  it('replaces the list on a second call (last-write-wins on the log)', async () => {
+    const ctx = await setup()
+    const agent = agentWithSession('writer-2')
+    await callTodo(ctx, { todos: [{ content: 'a', status: 'pending' }] }, { agent })
+    await callTodo(ctx, { todos: [
+      { content: 'a', status: 'completed' },
+      { content: 'b', status: 'in_progress' },
+    ] }, { agent })
+
+    const current = agent.session.events.findLast(e => e.type === 'todo/write')!.data.todos
+    expect(current).toEqual([
+      { content: 'a', status: 'completed' },
+      { content: 'b', status: 'in_progress' },
+    ])
+  })
+
+  it('rejects a malformed status before execute runs (registry arg-validation)', async () => {
+    const ctx = await setup()
+    const result = await callTodo(ctx, { todos: [{ content: 'x', status: 'doing' }] })
+    expect(result.isError).toBe(true)
+  })
+
+  it('rejects a non-array todos argument', async () => {
+    const ctx = await setup()
+    const result = await callTodo(ctx, { todos: 'nope' })
+    expect(result.isError).toBe(true)
+  })
+
+  it.each([
+    { label: 'empty content', todos: [{ content: '   ', status: 'pending' }], fragment: 'non-empty' },
+    { label: 'duplicate content', todos: [{ content: 'dup', status: 'pending' }, { content: 'dup', status: 'completed' }], fragment: 'duplicate' },
+    { label: 'two in_progress', todos: [{ content: 'a', status: 'in_progress' }, { content: 'b', status: 'in_progress' }], fragment: 'in_progress' },
+  ])('rejects $label as an isError result', async ({ todos, fragment }) => {
+    const ctx = await setup()
+    const result = await callTodo(ctx, { todos })
+    expect(result.isError).toBe(true)
+    expect(text(result)).toContain(fragment)
+  })
+
+  it('rejects a non-agent caller (the list has no owning session)', async () => {
+    const ctx = await setup()
+    const result = await callTodo(ctx, { todos: [{ content: 'a', status: 'pending' }] }, { agent: undefined })
+    expect(result.isError).toBe(true)
+    expect(text(result)).toContain('owning agent session')
+  })
+
+  it('presents the call with a stable title and the list as raw input', async () => {
+    const ctx = await setup()
+    const def = ctx.tools.get('todo_write')!
+    const todos = [{ content: 'a', status: 'pending' }]
+    expect(def.presentCall?.({ todos })).toEqual({ title: 'Update todo list', kind: 'other', rawInput: todos })
+  })
+
+  it('unregisters the tool when its contributing fiber is disposed (HMR-safety)', async () => {
+    const ctx = new Context()
+    await ctx.plugin(SystemPrompt)
+    await ctx.plugin(ToolRegistry)
+    const fiber = await ctx.plugin(tool)
+    expect(ctx.tools.schemas().some(s => s.name === 'todo_write')).toBe(true)
+    await fiber.dispose()
+    expect(ctx.tools.schemas().some(s => s.name === 'todo_write')).toBe(false)
+  })
+
+  it('has the namespace-plugin export shape (no stray default) so the Loader keeps name/inject/apply', () => {
+    // Postmortem 0001 guard: this plugin HAS `inject = ['tools']`, so a stray
+    // `export default apply` would collapse the module via `unwrapExports`
+    // (`exports.default ?? exports`), DROP `inject`, and crash at load with
+    // "cannot get property … without inject". Guard the shape directly.
+    expect('default' in tool).toBe(false)
+    expect(tool.name).toBe('tool-todo')
+    expect(tool.inject).toEqual(['tools'])
+
+    const loader = Object.create(Loader.prototype) as Loader
+    const unwrapped = loader.unwrapExports(tool) as Record<string, unknown>
+    expect(unwrapped).toBe(tool)
+    expect(unwrapped.name).toBe('tool-todo')
+    expect(unwrapped.inject).toEqual(['tools'])
+    expect(typeof unwrapped.apply).toBe('function')
+  })
+})

+ 27 - 0
packages/todo/tool-todo/tsconfig.json

@@ -0,0 +1,27 @@
+{
+  "extends": "../../../tsconfig.base.json",
+  "compilerOptions": {
+    "rootDir": "src",
+    "outDir": "lib/types"
+  },
+  "include": [
+    "src"
+  ],
+  "references": [
+    {
+      "path": "../../../vendor/cosmokit"
+    },
+    {
+      "path": "../../../vendor/cordis"
+    },
+    {
+      "path": "../../core/tools"
+    },
+    {
+      "path": "../../core/agent"
+    },
+    {
+      "path": "../../core/session"
+    }
+  ]
+}

+ 1 - 0
packages/ui/acp/package.json

@@ -44,6 +44,7 @@
     "@deepseek-ai/dsh-session-persistence-jsonl": "workspace:^",
     "@deepseek-ai/dsh-system-prompt": "workspace:^",
     "@deepseek-ai/dsh-tool-bash": "workspace:^",
+    "@deepseek-ai/dsh-tool-todo": "workspace:^",
     "@deepseek-ai/dsh-tools": "workspace:^",
     "cordis": "^4.0.0-rc.6"
   }

+ 19 - 1
packages/ui/acp/src/index.ts

@@ -53,6 +53,8 @@ import {
   type LoadSessionResponse,
   type NewSessionRequest,
   type NewSessionResponse,
+  type Plan,
+  type PlanEntry,
   type PromptRequest,
   type PromptResponse,
   type SessionNotification,
@@ -64,7 +66,7 @@ import { CallId } from '@deepseek-ai/dsh-llm'
 import type { Agent, AgentStatus } from '@deepseek-ai/dsh-agent'
 import { AgentId } from '@deepseek-ai/dsh-agent'
 import { SessionId } from '@deepseek-ai/dsh-session'
-import type { SessionEvent, TurnEndReason } from '@deepseek-ai/dsh-session'
+import type { SessionEvent, TodoItem, TurnEndReason } from '@deepseek-ai/dsh-session'
 import type { ToolCallKind, ToolCallPresentation, ToolRegistry, ToolResultPresentation, ToolTerminal } from '@deepseek-ai/dsh-tools'
 // Side-effect type import: declaration-merges `ctx.sessionPersistence` onto
 // Context (the bridge injects it and reads `list()` for load cwd validation).
@@ -873,6 +875,10 @@ export function streamSessionEventUpdate(
       })
       return
     }
+    case 'todo/write': {
+      notify({ sessionId, update: { sessionUpdate: 'plan', ...todosToPlan(event.data.todos) } })
+      return
+    }
     // turn/step boundaries, context/message, steering,
     // assistant/message — no direct ACP client update.
     default:
@@ -880,6 +886,18 @@ export function streamSessionEventUpdate(
   }
 }
 
+/**
+ * Map a harness todo list to an ACP `plan` body. ACP's `PlanEntry` requires
+ * `content` + `priority` + `status`, but a {@link TodoItem} carries no priority,
+ * so synthesize a constant `'medium'` on every entry; `status` maps 1:1 (the
+ * harness status triple IS `PlanEntryStatus`). The ACP client REPLACES its whole
+ * plan on each `plan` update, matching the harness's whole-list-replace
+ * semantics, so no per-entry diffing is needed.
+ */
+export function todosToPlan(todos: TodoItem[]): Plan {
+  return { entries: todos.map((todo): PlanEntry => ({ content: todo.content, priority: 'medium', status: todo.status })) }
+}
+
 /**
  * Per-connection terminal-rendering context threaded into
  * {@link streamSessionEventUpdate}: whether the client advertised the

+ 10 - 0
packages/ui/acp/tests/harness.ts

@@ -20,6 +20,7 @@ import AgentLoop from '@deepseek-ai/dsh-agent-loop'
 import SessionPersistenceJsonl from '@deepseek-ai/dsh-session-persistence-jsonl'
 import { LocalBashExecutor } from '@deepseek-ai/dsh-bash-local'
 import * as ToolBash from '@deepseek-ai/dsh-tool-bash'
+import * as ToolTodo from '@deepseek-ai/dsh-tool-todo'
 import {
   ClientSideConnection,
   ndJsonStream,
@@ -158,6 +159,12 @@ export async function makeBridgeHarness(options: {
    * implementation over a mock in tests").
    */
   withBash?: boolean
+  /**
+   * Plug the REAL `dsh-tool-todo` tool so a test can drive `todo_write` through
+   * the bridge and assert the resulting `plan` sessionUpdate — the shipping
+   * tool + the bridge's own todo/write→plan mapping, not a stand-in.
+   */
+  withTodo?: boolean
 } = { storageDir: '' }): Promise<BridgeHarness> {
   const adapter = new MockAdapter(options.script ?? [])
 
@@ -173,6 +180,9 @@ export async function makeBridgeHarness(options: {
     await ctx.plugin(LocalBashExecutor, { timeoutMs: 10_000 })
     await ctx.plugin(ToolBash)
   }
+  if (options.withTodo) {
+    await ctx.plugin(ToolTodo)
+  }
   ctx.llm.registerAdapter(['mock'], adapter)
 
   // Two identity byte pipes cross-wired into the two ndJsonStreams: bytes the

+ 38 - 0
packages/ui/acp/tests/load.spec.ts

@@ -94,6 +94,44 @@ describe('acp bridge — session/load replay', () => {
     expect(content[0]?.content.text).toBe('```console\nhello\n```')
   })
 
+  it('replays a persisted todo/write as a plan sessionUpdate on load', async () => {
+    // A turn whose model called todo_write persists a todo/write event. A fresh
+    // bridge loading the session must re-emit the ACP `plan` update from the log
+    // (the load replay runs every event through streamSessionEventUpdate), so an
+    // editor reopening the session sees the current plan.
+    live = await makeBridgeHarness({
+      storageDir,
+      withTodo: true,
+      script: [
+        toolCallResponse('c1', 'todo_write', {
+          todos: [
+            { content: 'first step', status: 'in_progress' },
+            { content: 'second step', status: 'pending' },
+          ],
+        }),
+        textResponse('planned'),
+      ],
+    })
+    await live.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: {} })
+    const { sessionId } = await live.client.newSession({ cwd: process.cwd(), mcpServers: [] })
+    await live.client.prompt({ sessionId, prompt: [{ type: 'text', text: 'plan it' }] })
+    await live.dispose()
+    live = undefined
+
+    loader = await makeBridgeHarness({ storageDir, withTodo: true, script: [] })
+    await loader.client.initialize({ protocolVersion: PROTOCOL_VERSION, clientCapabilities: {} })
+    await loader.client.loadSession({ sessionId, cwd: process.cwd(), mcpServers: [] })
+
+    const plan = loader.updates.find(u => u.sessionUpdate === 'plan')
+    expect(plan).toEqual({
+      sessionUpdate: 'plan',
+      entries: [
+        { content: 'first step', priority: 'medium', status: 'in_progress' },
+        { content: 'second step', priority: 'medium', status: 'pending' },
+      ],
+    })
+  })
+
   it('replays a persisted bash call as a TERMINAL card when the loader advertises the capability', async () => {
     // The presentation is resolved at replay time, so a loader that advertised
     // _meta.terminal_output must reconstruct the terminal card (content + _meta)

+ 38 - 1
packages/ui/acp/tests/stream-update.spec.ts

@@ -3,7 +3,7 @@ import { CallId } from '@deepseek-ai/dsh-llm'
 import { SessionId, type SessionEvent } from '@deepseek-ai/dsh-session'
 import type { SessionNotification } from '@agentclientprotocol/sdk'
 import type { ToolDefinition, ToolRegistry } from '@deepseek-ai/dsh-tools'
-import { streamSessionEventUpdate, agentOptions, ToolPresenter } from '../src/index.ts'
+import { streamSessionEventUpdate, agentOptions, todosToPlan, ToolPresenter } from '../src/index.ts'
 
 /** Collect the updates a single event produces (no presenter → generic fallback). */
 function updatesFor(event: SessionEvent): SessionNotification['update'][] {
@@ -118,6 +118,43 @@ describe('streamSessionEventUpdate', () => {
     expect(updatesFor(evt('turn/end', { turn: 1, reason: { kind: 'completed' } }))).toEqual([])
     expect(updatesFor(evt('step/start', { turn: 1, step: 1 }))).toEqual([])
   })
+
+  it('maps todo/write to a plan sessionUpdate with priority synthesized as medium', () => {
+    expect(updatesFor(evt('todo/write', {
+      todos: [
+        { content: 'plan the work', status: 'in_progress' },
+        { content: 'write the code', status: 'pending' },
+        { content: 'run the tests', status: 'completed' },
+      ],
+    }))).toEqual([{
+      sessionUpdate: 'plan',
+      entries: [
+        { content: 'plan the work', priority: 'medium', status: 'in_progress' },
+        { content: 'write the code', priority: 'medium', status: 'pending' },
+        { content: 'run the tests', priority: 'medium', status: 'completed' },
+      ],
+    }])
+  })
+
+  it('maps an empty todo list to a plan with no entries', () => {
+    expect(updatesFor(evt('todo/write', { todos: [] }))).toEqual([{ sessionUpdate: 'plan', entries: [] }])
+  })
+})
+
+describe('todosToPlan', () => {
+  it('maps status 1:1 and stamps every entry priority medium', () => {
+    expect(todosToPlan([
+      { content: 'a', status: 'pending' },
+      { content: 'b', status: 'in_progress' },
+      { content: 'c', status: 'completed' },
+    ])).toEqual({
+      entries: [
+        { content: 'a', priority: 'medium', status: 'pending' },
+        { content: 'b', priority: 'medium', status: 'in_progress' },
+        { content: 'c', priority: 'medium', status: 'completed' },
+      ],
+    })
+  })
 })
 
 describe('ToolPresenter (tool-owned presentation via the tool registry)', () => {

+ 27 - 0
pnpm-lock.yaml

@@ -624,6 +624,30 @@ importers:
         specifier: ^4.0.0-rc.6
         version: 4.0.0-rc.6(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.4)
 
+  packages/todo/tool-todo:
+    devDependencies:
+      '@deepseek-ai/dsh-agent':
+        specifier: workspace:^
+        version: link:../../core/agent
+      '@deepseek-ai/dsh-agent-loop':
+        specifier: workspace:^
+        version: link:../../core/agent-loop
+      '@deepseek-ai/dsh-llm':
+        specifier: workspace:^
+        version: link:../../llm/llm
+      '@deepseek-ai/dsh-session':
+        specifier: workspace:^
+        version: link:../../core/session
+      '@deepseek-ai/dsh-system-prompt':
+        specifier: workspace:^
+        version: link:../../core/system-prompt
+      '@deepseek-ai/dsh-tools':
+        specifier: workspace:^
+        version: link:../../core/tools
+      cordis:
+        specifier: ^4.0.0-rc.6
+        version: 4.0.0-rc.6(@cordisjs/plugin-include@1.0.4)(@cordisjs/plugin-loader@1.0.0-rc.4)
+
   packages/ui/acp:
     dependencies:
       '@agentclientprotocol/sdk':
@@ -663,6 +687,9 @@ importers:
       '@deepseek-ai/dsh-tool-bash':
         specifier: workspace:^
         version: link:../../bash/tool-bash
+      '@deepseek-ai/dsh-tool-todo':
+        specifier: workspace:^
+        version: link:../../todo/tool-todo
       '@deepseek-ai/dsh-tools':
         specifier: workspace:^
         version: link:../../core/tools

+ 2 - 2
scripts/gen-cordis-catalog.ts

@@ -394,7 +394,7 @@ function render(events: EventEntry[], services: ServiceEntry[]): string {
     '',
     '## Events',
     '',
-    `Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets \`next()\` and may transform or veto — see [waterfall semantics](../architecture.md#cordis-waterfall-semantics-important)), **parallel** (awaited fan-out, no veto), **serial** (awaited, in registration order, no veto). The harness declares ${events.length} events across ${new Set(events.map(e => e.scope)).size} scopes.`,
+    'Dispatch modes: **emit** (fire-and-forget), **waterfall** (each listener gets `next()` and may transform or veto — see [waterfall semantics](../architecture.md#cordis-waterfall-semantics-important)), **parallel** (awaited fan-out, no veto), **serial** (awaited, in registration order, no veto).',
     '',
   ]
   const scopes = [...new Set(events.map(e => e.scope))].sort()
@@ -407,7 +407,7 @@ function render(events: EventEntry[], services: ServiceEntry[]): string {
   lines.push(
     '## Services',
     '',
-    `The ${services.length} \`ctx.<key>\` services the harness provides. An abstract seam (e.g. \`ctx.bash\`) is implemented by a separate package; the interface is what consumers code against.`,
+    'The `ctx.<key>` services the harness provides. An abstract seam (e.g. `ctx.bash`) is implemented by a separate package; the interface is what consumers code against.',
     '',
   )
   for (const s of services) lines.push(...renderService(s))

+ 1 - 0
scripts/type-equiv.manifest.json

@@ -16,6 +16,7 @@
     { "doc": "docs/core-data-structures/llm-streaming.md", "symbol": "ContentBlockMap", "source": "packages/llm/llm/src/types.ts" },
 
     { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEventMap", "source": "packages/core/session/src/types.ts" },
+    { "doc": "docs/core-data-structures/session.md", "symbol": "TodoItem", "source": "packages/core/session/src/types.ts" },
     { "doc": "docs/core-data-structures/session.md", "symbol": "SessionEvent", "source": "packages/core/session/src/types.ts" },
     { "doc": "docs/core-data-structures/session.md", "symbol": "TurnTriggerMap", "source": "packages/core/session/src/types.ts" },
     { "doc": "docs/core-data-structures/session.md", "symbol": "TurnEndReasonMap", "source": "packages/core/session/src/types.ts" },

+ 1 - 0
tsconfig.base.json

@@ -45,6 +45,7 @@
         "./packages/bash/*/src",
         "./packages/compact/*/src",
         "./packages/subagent/*/src",
+        "./packages/todo/*/src",
         "./packages/session-persistence/*/src",
         "./packages/ui/*/src",
         "./packages/util/*/src",

+ 2 - 1
tsconfig.build.json

@@ -40,6 +40,7 @@
     { "path": "./packages/subagent/subagent-inprocess" },
     { "path": "./packages/subagent/subagent-spawn" },
     { "path": "./packages/subagent/subagent-fork" },
-    { "path": "./packages/subagent/subagent-acp" }
+    { "path": "./packages/subagent/subagent-acp" },
+    { "path": "./packages/todo/tool-todo" }
   ]
 }

+ 2 - 1
tsconfig.json

@@ -51,6 +51,7 @@
     { "path": "./packages/subagent/subagent-inprocess" },
     { "path": "./packages/subagent/subagent-spawn" },
     { "path": "./packages/subagent/subagent-fork" },
-    { "path": "./packages/subagent/subagent-acp" }
+    { "path": "./packages/subagent/subagent-acp" },
+    { "path": "./packages/todo/tool-todo" }
   ]
 }