Răsfoiți Sursa

feat(session, llm): durable image offload watermark

Record the image offload point as the core image/offload session event so the
derived surface, every adapter, and the token meter share one offloaded set.
The agent loop plans the advance from the route's declared request-image
budget before dispatch; an adapter whose exact accounting still overflows
fails with IMAGE_OFFLOAD_REQUIRED and the loop advances by the named count.
creatixchu 3 săptămâni în urmă
părinte
comite
345b5cdc6f
86 a modificat fișierele cu 1955 adăugiri și 624 ștergeri
  1. 6 0
      .agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.i18n.yaml
  2. 53 0
      .agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.md
  3. 53 0
      .agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.zh.md
  4. 0 6
      .agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.i18n.yaml
  5. 0 60
      .agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.md
  6. 0 60
      .agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.zh.md
  7. 2 2
      docs/architecture.i18n.yaml
  8. 5 2
      docs/architecture.md
  9. 5 2
      docs/architecture.zh.md
  10. 2 2
      docs/event-producer-consumer.i18n.yaml
  11. 5 5
      docs/event-producer-consumer.md
  12. 5 5
      docs/event-producer-consumer.zh.md
  13. 2 2
      docs/persistence-catalog.i18n.yaml
  14. 35 13
      docs/persistence-catalog.md
  15. 22 0
      docs/persistence-catalog.zh.md
  16. 2 2
      docs/subsystems/core.i18n.yaml
  17. 1 1
      docs/subsystems/core.md
  18. 1 1
      docs/subsystems/core.zh.md
  19. 2 2
      docs/subsystems/llm-streaming.i18n.yaml
  20. 15 3
      docs/subsystems/llm-streaming.md
  21. 15 3
      docs/subsystems/llm-streaming.zh.md
  22. 2 2
      docs/subsystems/session.i18n.yaml
  23. 41 1
      docs/subsystems/session.md
  24. 41 1
      docs/subsystems/session.zh.md
  25. 2 2
      packages/core/agent-loop/README.i18n.yaml
  26. 1 1
      packages/core/agent-loop/README.md
  27. 1 1
      packages/core/agent-loop/README.zh.md
  28. 60 6
      packages/core/agent-loop/src/agent.ts
  29. 161 0
      packages/core/agent-loop/tests/image-offload.spec.ts
  30. 3 1
      packages/core/agent-loop/tests/mock-adapter.ts
  31. 2 2
      packages/core/session/README.i18n.yaml
  32. 4 1
      packages/core/session/README.md
  33. 4 1
      packages/core/session/README.zh.md
  34. 98 0
      packages/core/session/src/image-offload.ts
  35. 43 4
      packages/core/session/src/index.ts
  36. 2 1
      packages/core/session/src/invariant.ts
  37. 1 0
      packages/core/session/src/known-event-types.ts
  38. 25 0
      packages/core/session/src/types.ts
  39. 161 0
      packages/core/session/tests/image-offload.spec.ts
  40. 2 2
      packages/llm/llm-deepseek/README.i18n.yaml
  41. 2 2
      packages/llm/llm-deepseek/README.md
  42. 2 2
      packages/llm/llm-deepseek/README.zh.md
  43. 25 16
      packages/llm/llm-deepseek/src/adapter.ts
  44. 26 39
      packages/llm/llm-deepseek/src/request-pricing.ts
  45. 43 19
      packages/llm/llm-deepseek/src/serialize.ts
  46. 33 15
      packages/llm/llm-deepseek/tests/adapter.spec.ts
  47. 20 2
      packages/llm/llm-deepseek/tests/dynamic-config.spec.ts
  48. 18 21
      packages/llm/llm-deepseek/tests/request-pricing.spec.ts
  49. 22 6
      packages/llm/llm-deepseek/tests/serialize.spec.ts
  50. 2 2
      packages/llm/llm-pi-ai/README.i18n.yaml
  51. 3 3
      packages/llm/llm-pi-ai/README.md
  52. 3 3
      packages/llm/llm-pi-ai/README.zh.md
  53. 9 0
      packages/llm/llm-pi-ai/src/adapter.ts
  54. 35 23
      packages/llm/llm-pi-ai/src/context.ts
  55. 28 26
      packages/llm/llm-pi-ai/tests/context.spec.ts
  56. 2 2
      packages/llm/llm/README.i18n.yaml
  57. 3 3
      packages/llm/llm/README.md
  58. 3 3
      packages/llm/llm/README.zh.md
  59. 4 1
      packages/llm/llm/src/adapter-failure.ts
  60. 173 89
      packages/llm/llm/src/content.ts
  61. 9 0
      packages/llm/llm/src/error.ts
  62. 42 0
      packages/llm/llm/src/index.ts
  63. 42 2
      packages/llm/llm/src/types.ts
  64. 100 114
      packages/llm/llm/tests/content.spec.ts
  65. 30 0
      packages/llm/llm/tests/service.spec.ts
  66. 2 2
      packages/llm/token-meter/README.i18n.yaml
  67. 1 1
      packages/llm/token-meter/README.md
  68. 1 1
      packages/llm/token-meter/README.zh.md
  69. 15 2
      packages/llm/token-meter/src/index.ts
  70. 15 3
      packages/llm/token-meter/src/route-pricing.ts
  71. 18 13
      packages/llm/token-meter/src/surface-fold.ts
  72. 26 0
      packages/llm/token-meter/tests/route-pricing.spec.ts
  73. 2 2
      packages/test-support/llm-replay/README.i18n.yaml
  74. 1 1
      packages/test-support/llm-replay/README.md
  75. 1 1
      packages/test-support/llm-replay/README.zh.md
  76. 28 5
      packages/test-support/llm-replay/src/index.ts
  77. 38 3
      packages/test-support/llm-replay/tests/llm-replay.spec.ts
  78. 5 0
      scripts/type-equiv.manifest.json
  79. 5 0
      snapshots/acp/acp.snapshot.ts
  80. 56 0
      snapshots/acp/image-offload/cordis.snapshot.yml
  81. 33 0
      snapshots/acp/image-offload/cordis.yml
  82. 86 0
      snapshots/acp/image-offload/input.json
  83. 33 0
      snapshots/acp/image-offload/session.jsonl
  84. 16 0
      snapshots/acp/image-offload/snapshot.yml
  85. 8 0
      snapshots/acp/image-offload/stdout.expected.jsonl
  86. 1 0
      snapshots/acp/image-offload/system-prompt.expected.md

+ 6 - 0
.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.i18n.yaml

@@ -0,0 +1,6 @@
+# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each
+# side as of the last confirmed-consistent state. Both languages carry equal authority;
+# after editing either side, bring the other along and re-record with:
+#   pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.md
+2026-09-02-image-offload-watermark.md: b525554a31100a19f2ff27df178dc7c5b1b8d392
+2026-09-02-image-offload-watermark.zh.md: abe44a9c18d619ca877a4adfe33bef7017913338

+ 53 - 0
.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.md

@@ -0,0 +1,53 @@
+# Agent Note: Durable image offload watermark
+
+Status: implemented
+
+English | [中文](2026-09-02-image-offload-watermark.zh.md)
+
+## Problem
+
+Request-size image offload was recomputed from scratch on every request. Each route collected every image occurrence on the derived surface, oldest first, and once the accumulated bytes exceeded its budget it rounded the excess up to a whole removal quantum and replaced that many oldest occurrences with placeholder text, per the [unified image request pipeline](../feature/2026-08-20-unified-image-request-pipeline.md). Nothing remembered where the previous request stopped; the prefix was stable only because the arithmetic over an append-only history repeated itself.
+
+That stability failed wherever the arithmetic inputs moved. The [Files inline fallback](../bug-fix/2026-08-21-deepseek-files-inline-fallback.md) rebuilt a request under a 20 MiB inline budget with a 10 MiB quantum, offloading far more images for that one request, and the next request in file mode brought them back. The pi-ai route used a quantum of one byte, so its prefix moved on almost every request. Compaction lowered the total and returned previously offloaded images. A route switch moved every step boundary. Each move changed the model-visible prefix and invalidated the provider cache prefix.
+
+The same recomputation broke the repository invariant that model-visible input is reconstructable from the session log. Which representation was dispatched, the exact derived request-version byte lengths, and the route budgets and quanta were runtime or configuration facts that never entered the log; `request/header` records call config, system prompt, and tools only. Provider usage anchors token totals and cannot recover the image set, and the [route-priced estimate](../feature/2026-08-24-route-priced-image-request-pressure.md) documented that it did not reproduce the fallback budget. No consumer could pair a logged assistant response with the image set its request carried.
+
+## Decision
+
+The offload point is a durable session fact: the core `image/offload` event records an image offload watermark that only advances, and the derived surface, every route, and the token meter read the offloaded set from it.
+
+**Event.** `image/offload` carries `{ turn, step, watermark }`, where `watermark` is an `ImageOccurrencePosition`: the seq of the event carrying the last offloaded occurrence and its block path inside that event's content (the top-level block index, then the index inside a tool-result block). Positions order by seq, then by path, so a newer event always lies after an older one regardless of surface replacements. `Session.deriveMessages()` marks every occurrence positioned at or before the watermark `offloaded: true` on the `ImageBlock`; the marked copies are frozen and the durable event content is untouched. `Session.append` and seeding reject a watermark that is malformed, names an event outside the log, or does not advance strictly past the previous one; `session.imageOffloadWatermark()` folds the latest. The event is required-on-read because it changes the derived surface; `SESSION_FORMAT_VERSION` is unchanged because the log structure is not.
+
+**Only advances.** The watermark never retreats when a budget grows, a route changes, or compaction lowers the total, so the model-visible prefix and the provider cache prefix move only forward. An occurrence below the watermark that compaction later shadows leaves the watermark valid, because the comparison is positional.
+
+**Decision owned by the loop, budgets declared by routes.** An image-capable route declares an `LlmImageRequestBudget` (`representation`, `maxBytes`, `maxImages`, both quanta, and the request-version byte target) as `imageRequest` on its `LlmResolvedModelInfo`; `LlmRuntime` validates it and exposes it on `PreparedLlmCall`. After `request/header`, `buildRequest` collects the retained occurrences from the surface in log order, plans the advance with the pure `planImageOffload()` (represented bytes are the normalized count clamped to the version target and base64-expanded for inline routes, removed in whole quanta), appends the event, and only then derives the request messages. The DeepSeek adapter declares its file-mode budget; the pi-ai adapter declares its base64 bound; the replay adapter declares an optional `imageRequestMaxBytes` for keyless scenarios.
+
+**Adapters project, never decide.** Serialization renders every `offloaded` block as `offloadedImageText` with the currently resolved access path and prepares only retained occurrences. When the retained occurrences' exact request-version bytes still exceed the route budget, in file mode, under the inline fallback's tighter budget, or under the pi-ai bound, the adapter fails the attempt with `IMAGE_OFFLOAD_REQUIRED` and `LlmFailure.offloadImages` naming how many more oldest occurrences must be offloaded, computed with `offloadedImagePrefixCount()`. The loop advances the watermark by that count and rebuilds the request before `agent/request-error` runs; when nothing remains to offload the failure reaches ordinary recovery.
+
+**Token accounting.** `priceImages` receives surface `ImageBlock`s and prices an `offloaded` one as its placeholder text; the DeepSeek and replay pricing no longer reproduce any offload arithmetic. The meter folds `image/offload` into its replay state, prices the current surface under the current watermark and each usage anchor under the watermark its request was derived with. Provider usage remains the anchor for completed requests.
+
+**Other consumers.** Compaction summarization and every other `ctx.llm.stream` caller derive from the same surface, so they read the watermark and never advance it. Resume, fork, and replay reproduce the surface from the log. Text-only routes keep their separate whole-history substitution.
+
+## Alternatives considered
+
+**Keep recomputing the offload point per request.** Stable only while the arithmetic inputs held still; the inline fallback, the pi-ai quantum, compaction, and route switches all moved the prefix, and no consumer could reconstruct a historical request's image set.
+
+**Record each request's projection outcome as a log-only event.** Restores reconstructability but not stability: the recorded outcome is not a decision input, so every oscillation still happens and the log merely documents it, with two sources of truth for the offloaded set that can disagree.
+
+**Log the full projected request body.** Everything except the offload decision is already derivable; repeating the history per request grows the log quadratically to record one position.
+
+**Identify the watermark by attachment id or by occurrence count.** Attachment ids repeat for duplicate attachments, so the position is ambiguous; counts shift when compaction prunes earlier occurrences. A seq plus block path is unambiguous and survives pruning.
+
+**Let each adapter append the event.** The adapter owns the budgets but not the session surface; a surface fact appended below the loop bypasses the loop's ownership of derived history and lets two adapters define the surface differently. Adapters report the count they need instead.
+
+**Keep a transient extra offload for the inline fallback and exact-byte overflow.** Would have sent an unlogged projection in exactly the cases the invariant exists for; the failure-and-advance path costs one serialization attempt and keeps every dispatched request derivable from the log.
+
+## Consequences
+
+Offloaded images never return automatically when budgets grow, a larger route is selected, or compaction lowers the total; recovery is the read-only path in the placeholder, which the model uses deliberately. A Files outage or a temporary switch to a small-budget route advances the watermark permanently; both are accepted for the same reason.
+
+Every dispatched request's image set is determined by the log alone, across file mode, inline fallback, resume, fork, retry, and compaction, and the provider cache prefix no longer oscillates. The execution-world access path embedded in placeholder and handle text is still resolved at serialization time; that gap exists for retained images too and belongs to a separate decision about recording the execution-world mapping.
+
+## Testing
+
+`packages/llm/llm/tests/content.spec.ts` pins the image walk, position ordering, represented bytes, marking, projection, and the watermark planner, including the 129-to-64 MiB quantum example. `packages/core/session/tests/image-offload.spec.ts` pins append and seed validation, strict advance, nested marking, frozen copies, cache rebuild, and scratch-replay equality. `packages/core/agent-loop/tests/image-offload.spec.ts` pins the pre-dispatch advance, the non-retreating watermark, the `IMAGE_OFFLOAD_REQUIRED` advance-and-rebuild path, and the exhausted case. Adapter specs pin placeholder projection, prepared-only-retained reads, and the exact-byte failure with its count; `route-pricing.spec.ts` pins watermark pricing; the replay adapter spec pins `imageRequestMaxBytes`. The `image-offload` ACP snapshot replays an authored six-frame session through the shipped profile under a replay route whose base64 budget admits three frames, and pins the `image/offload` watermark appended after the first `request/header` and the unchanged second turn.

+ 53 - 0
.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.zh.md

@@ -0,0 +1,53 @@
+# Agent Note: 持久的图片 offload 水位
+
+Status: implemented
+
+[English](2026-09-02-image-offload-watermark.md) | 中文
+
+## 问题
+
+请求级图片 offload 过去在每次请求时都从头重算。每条路由按最老优先的顺序收集派生表层上的全部图片出现位置,累计字节一旦超过预算,就把超出部分向上取整到整个删除量子,把这么多最老的出现位置替换为占位文本,见[统一图片请求管线](../feature/2026-08-20-unified-image-request-pipeline.zh.md)。没有任何东西记住上一次请求停在哪里,前缀稳定只是因为对 append-only 历史做同样的算术会得到同样的结果。
+
+算术的输入一动,这种稳定就失效。[Files 内联回退](../bug-fix/2026-08-21-deepseek-files-inline-fallback.zh.md)会用 20 MiB 内联预算和 10 MiB 量子重建请求,那一次省略多得多的图片,下一次 file 模式的请求又把它们带回来。pi-ai 路由的量子是一个字节,前缀几乎每次请求都会移动。compaction 降低总量,让先前省略的图片回归。切换路由会移动每个台阶边界。每一次移动都改变模型可见前缀,并让 provider 的缓存前缀失效。
+
+同样的重算也破坏了仓库不变量:模型可见输入必须能从 session log 重建。实际发出的表示方式、派生请求版本的精确字节长度、路由预算和量子都是运行时或配置事实,从不进入日志,`request/header` 只记录调用配置、系统提示词和工具。provider usage 只锚定 token 总量,恢复不了图片集合,[按路由定价的估计](../feature/2026-08-24-route-priced-image-request-pressure.zh.md)也写明它不复现回退预算。没有任何消费方能把一条已记录的助手响应和它的请求携带的图片集合配对。
+
+## 决定
+
+offload 位置是持久的会话事实:核心事件 `image/offload` 记录一条只会前进的图片 offload 水位,派生表层、每条路由和 token meter 都从它读取省略集合。
+
+**事件。** `image/offload` 携带 `{ turn, step, watermark }`,其中 `watermark` 是一个 `ImageOccurrencePosition`:承载最后一个被省略出现位置的事件序号,以及它在该事件内容中的块路径(顶层块下标,再是工具结果块内的下标)。位置先按序号再按路径排序,所以无论表层如何替换,较新的事件总在较老的之后。`Session.deriveMessages()` 把位于水位及之前的每个出现位置在 `ImageBlock` 上标为 `offloaded: true`;标记后的副本被冻结,持久事件内容不受影响。`Session.append` 与 seed 拒绝畸形、指向日志之外事件或没有严格越过前一条的水位;`session.imageOffloadWatermark()` 折叠最新值。该事件改变派生表层,因此读取时必须识别;日志结构没有变化,`SESSION_FORMAT_VERSION` 保持不变。
+
+**只前进。** 预算变大、路由切换或 compaction 降低总量时水位永不回退,所以模型可见前缀和 provider 缓存前缀只向前移动。水位之下的出现位置后来被 compaction 遮蔽也不影响水位有效性,因为比较是按位置进行的。
+
+**决定权在循环,预算由路由声明。** 支持图片的路由在其 `LlmResolvedModelInfo` 上以 `imageRequest` 声明一个 `LlmImageRequestBudget`(`representation`、`maxBytes`、`maxImages`、两个量子与请求版本字节目标);`LlmRuntime` 校验它并通过 `PreparedLlmCall` 暴露。`request/header` 之后,`buildRequest` 按日志顺序从表层收集保留的出现位置,用纯函数 `planImageOffload()` 规划推进(表示字节是归一化字节数按版本目标截断后的值,内联路由再按 base64 展开,按整量子删除),追加事件,然后才派生请求消息。DeepSeek adapter 声明其 file 模式预算,pi-ai adapter 声明其 base64 上限,replay adapter 为 keyless 场景声明可选的 `imageRequestMaxBytes`。
+
+**adapter 只投影,不决定。** 序列化把每个 `offloaded` 块渲染为带当前已解析访问路径的 `offloadedImageText`,只准备保留的出现位置。当保留的出现位置按精确请求版本字节仍超过路由预算,无论是 file 模式、内联回退更紧的预算还是 pi-ai 上限,adapter 都以 `IMAGE_OFFLOAD_REQUIRED` 让本次尝试失败,并在 `LlmFailure.offloadImages` 中用 `offloadedImagePrefixCount()` 算出还需省略多少最老的出现位置。循环按该数量推进水位并在 `agent/request-error` 运行前重建请求;没有可省略的出现位置时,失败进入普通恢复路径。
+
+**token 记账。** `priceImages` 接收表层的 `ImageBlock`,把 `offloaded` 的按占位文本定价;DeepSeek 和 replay 的定价不再复现任何 offload 算术。meter 把 `image/offload` 折进其重放状态,按当前水位为当前表层定价,按每个 usage 锚点的请求派生时的水位为该锚点定价。已完成请求仍以 provider usage 为锚点。
+
+**其他消费方。** compaction 摘要和其他所有 `ctx.llm.stream` 调用方都从同一表层派生,因此只读取水位、从不推进。resume、fork 和重放从日志复现表层。纯文本路由保留各自的全历史替换。
+
+## 考虑过的替代方案
+
+**继续每次请求重算 offload 位置。** 只在算术输入不动时稳定,内联回退、pi-ai 量子、compaction 和路由切换都会移动前缀,且没有消费方能重建历史请求的图片集合。
+
+**用 log-only 事件记录每次请求的投影结果。** 恢复了可重建性但没有稳定性:记录的结果不是决策输入,每一种抖动照旧发生,日志只是把它记下来,且省略集合有两个可能不一致的事实来源。
+
+**记录完整的投影后请求体。** 除 offload 决定外一切都已可派生,为记录一个位置而每次请求重复整段历史会让日志平方级增长。
+
+**用附件 id 或出现次数标识水位。** 附件 id 在重复附加时重复,位置因此含糊;compaction 剪掉更早的出现位置后计数会漂移。序号加块路径没有歧义,也不受剪枝影响。
+
+**让各个 adapter 自己追加事件。** adapter 拥有预算,但不拥有会话表层;在循环之下追加表层事实绕过了循环对派生历史的所有权,也会让两个 adapter 对表层做出不同定义。adapter 改为上报它需要的数量。
+
+**为内联回退和精确字节溢出保留临时的额外省略。** 恰好会在不变量所针对的场景发送未记录的投影;失败再推进的路径只多花一次序列化尝试,且让每个已发出请求都可由日志派生。
+
+## 后果
+
+预算变大、选中更大的路由或 compaction 降低总量时,被省略的图片不会自动回归;恢复手段是占位文本中的只读路径,模型需要时主动使用。一次 Files 故障或临时切到小预算路由会永久推进水位,两者出于同一理由被接受。
+
+每个已发出请求的图片集合仅由日志决定,覆盖 file 模式、内联回退、resume、fork、retry 和 compaction,provider 缓存前缀不再抖动。占位和句柄文本中嵌入的执行世界访问路径仍在序列化时解析,这个缺口对保留的图片同样存在,属于另一个关于记录执行世界映射的决定。
+
+## 测试
+
+`packages/llm/llm/tests/content.spec.ts` 钉住图片遍历、位置排序、表示字节、标记、投影和水位规划器,包括 129 到 64 MiB 的量子示例。`packages/core/session/tests/image-offload.spec.ts` 钉住追加与 seed 校验、严格推进、嵌套标记、冻结副本、缓存重建和从头重放的一致性。`packages/core/agent-loop/tests/image-offload.spec.ts` 钉住发送前推进、不回退的水位、`IMAGE_OFFLOAD_REQUIRED` 的推进并重建路径以及耗尽的情况。adapter 测试钉住占位投影、只读取保留图片以及带数量的精确字节失败;`route-pricing.spec.ts` 钉住水位定价;replay adapter 测试钉住 `imageRequestMaxBytes`。`image-offload` ACP 快照通过发布的 profile、在一条 base64 预算只容纳三帧的 replay 路由下重放一段人工编写的六帧会话,钉住第一次 `request/header` 之后追加的 `image/offload` 水位和保持不变的第二轮。

+ 0 - 6
.agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.i18n.yaml

@@ -1,6 +0,0 @@
-# Bilingual-pair consistency record (docs/i18n/README.md): the git blob hash of each
-# side as of the last confirmed-consistent state. Both languages carry equal authority;
-# after editing either side, bring the other along and re-record with:
-#   pnpm run verify-translation-pairing --write .agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.md
-2026-09-02-image-offload-watermark.md: 61864bb8fd81af0ac0e3668b0c0ebc872ac1c031
-2026-09-02-image-offload-watermark.zh.md: 7fb9503ad8de8e02f358ada598bbb54f2a493445

+ 0 - 60
.agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.md

@@ -1,60 +0,0 @@
-# Agent Note: Durable image offload watermark
-
-Status: proposed
-
-English | [中文](2026-09-02-image-offload-watermark.zh.md)
-
-## Problem
-
-Request-size image offload is recomputed from scratch on every request. Each route collects every image occurrence on the derived surface, oldest first, and once the accumulated bytes exceed its budget it rounds the excess up to a whole removal quantum and replaces that many oldest occurrences with placeholder text, per the [unified image request pipeline](../../implemented/feature/2026-08-20-unified-image-request-pipeline.md). Under the DeepSeek defaults of a 128 MiB budget and a 64 MiB quantum, 129 one-mebibyte images offload the oldest 65, and that prefix holds until history passes 192 MiB. Nothing remembers where the previous request stopped; the prefix is stable only because the arithmetic over an append-only history repeats itself.
-
-That stability fails wherever the arithmetic inputs move. The [Files inline fallback](../../implemented/bug-fix/2026-08-21-deepseek-files-inline-fallback.md) rebuilds a request under a 20 MiB inline budget with a 10 MiB quantum, offloading far more images for that one request, and the next request in file mode brings them back. The pi-ai route uses a quantum of one byte, so its prefix moves on almost every request. Compaction lowers the total and returns previously offloaded images. A route switch moves every step boundary. Each move changes the model-visible prefix and invalidates the provider cache prefix.
-
-The same recomputation breaks the repository invariant that model-visible input is reconstructable from the session log. Which representation was dispatched, the exact derived request-version byte lengths, and the route budgets and quanta are runtime or configuration facts that never enter the log; `request/header` records call config, system prompt, and tools only. Provider usage anchors token totals and cannot recover the image set, and the [route-priced estimate](../../implemented/feature/2026-08-24-route-priced-image-request-pressure.md) documents that it does not reproduce the fallback budget. No consumer can pair a logged assistant response with the image set its request carried.
-
-## Proposal
-
-Make the offload point a durable session fact: an image offload watermark recorded as a surface-changing session event that only advances.
-
-**Event.** A new required-on-read `SessionEventMap` member (working name `image/offload`) carrying `{ turn, step }` and the watermark position. It is a surface event in the same sense as the compaction events: derived history marks every image occurrence positioned before the watermark as offloaded, and every route projects a marked occurrence as its `offloadedImageText` placeholder. Occurrences at or after the watermark are retained. Adapters keep their existing placeholder rendering; the only change is that the offloaded set comes from the surface rather than from per-request arithmetic.
-
-**Only advances.** A route whose budget is exceeded advances the watermark; a route with headroom leaves it alone. The watermark never retreats when a budget grows, a route changes, or compaction lowers the total, so the model-visible prefix and the provider cache prefix move only forward. Advancement keeps the existing count and byte quanta, so one advance still jumps to the next removal boundary rather than to the minimum that fits.
-
-**Position by occurrence.** The watermark names the first retained image occurrence by the sequence number of the event that carries it and the block path inside that event's content, including nested tool-result content. It does not use attachment ids, which are content digests that repeat when one image is attached twice, and it does not count occurrences, which compaction pruning would shift.
-
-**Appended before dispatch.** Like `request/header`, the event is appended inside its step before the request is sent, so the projection of every dispatched request is fully determined by the log at dispatch time. When the file-to-inline fallback needs a further advance, the adapter reports the requirement before the inline dispatch and the loop appends a second advance in the same step; the fallback still never sends an unlogged projection.
-
-**Never ignorable.** The event changes the derived surface, so a build that does not know the type must refuse the log rather than replay it with images the model never saw. `SESSION_FORMAT_VERSION` stays unchanged because the log structure does not change.
-
-**Decision owned by the loop.** `LlmRuntime` computes the watermark from budgets the route declares — representation, byte and count budgets, quanta — and appends the event; adapters declare budgets and report fallback requirements but do not decide the offloaded set. This makes the route-capability metadata seam that #2644 deferred a prerequisite of the implementation.
-
-**Consequences for other consumers.** The token meter prices the exact offloaded set instead of reproducing first-stage arithmetic. Resume, fork, and replay reproduce the surface from the log. Compaction that prunes messages below the watermark leaves it valid because the position names a retained occurrence. Text-only routes keep their separate whole-history substitution and do not touch the watermark.
-
-**Out of scope.** The execution-world access path embedded in placeholder and handle text is still resolved at serialization time; that gap exists for retained images too and belongs to a separate decision about recording the execution-world mapping. Recording per-request projection outcomes as a log-only event is unnecessary once the surface owns the offloaded set.
-
-## Alternatives considered
-
-**Keep recomputing the offload point per request (status quo).** Stable only while the arithmetic inputs hold still; the inline fallback, the pi-ai quantum, compaction, and route switches all move the prefix, and no consumer can reconstruct a historical request's image set. This is the state the issue asks to resolve.
-
-**Record each request's projection outcome as a log-only event.** Restores reconstructability but not stability: the recorded outcome is not a decision input, so every oscillation above still happens and the log merely documents it. It also creates two sources of truth for the offloaded set, the surface arithmetic and the record, which can disagree.
-
-**Log the full projected request body.** Everything except the offload decision is already derivable; repeating the history per request grows the log quadratically to record one position.
-
-**Identify the watermark by attachment id or by occurrence count.** Attachment ids repeat for duplicate attachments, so the position is ambiguous; counts shift when compaction prunes earlier occurrences. A sequence number plus block path is unambiguous and survives pruning.
-
-**Let each adapter append the event.** The adapter owns the budgets but not the session surface; a surface event appended below the loop bypasses the loop's ownership of derived history and would let two adapters define the surface differently.
-
-## Acceptance criteria
-
-- The watermark advances under file-mode budget pressure, inline fallback, and a route switch to a smaller budget, and never retreats after compaction, a budget increase, or a switch back.
-- From the log alone, the offloaded set of every dispatched request is determined, including the fallback's second advance, across resume, fork, retry, and compaction.
-- Replay, resume, and fork derive an identical surface, and a build without the event type refuses the log.
-- Keyless recorded-session snapshots cover file-mode offload, inline-fallback advance, and compaction below the watermark; unit tests pin the position encoding, monotonicity, and pre-dispatch append timing; both SDK expected outputs update for the new surface event.
-- The token meter's image pricing consumes the surface's offloaded set and the first-stage arithmetic reproduction is removed.
-
-## Risks
-
-- Offloaded images never return automatically when budgets grow; recovery is the read-only path in the placeholder, which the model must act on deliberately.
-- The loop-side decision requires routes to declare budgets through a capability metadata seam that does not exist yet; the implementation cannot land without it.
-- The fallback's pre-dispatch second advance adds an adapter-to-loop channel; without a stated scope it invites other transient request facts into surface events.
-- The execution-world path in placeholder text remains unlogged; this proposal narrows the reconstruction gap to that one input rather than closing it.

+ 0 - 60
.agents/notes/proposed/architecture/2026-09-02-image-offload-watermark.zh.md

@@ -1,60 +0,0 @@
-# Agent Note: 持久的图片 offload 水位
-
-Status: proposed
-
-[English](2026-09-02-image-offload-watermark.md) | 中文
-
-## 问题
-
-请求级图片 offload 在每次请求时都从头重算。每条路由按最老优先的顺序收集派生表层上的全部图片出现位置,累计字节一旦超过预算,就把超出部分向上取整到整个删除量子,把这么多最老的出现位置替换为占位文本,见[统一图片请求管线](../../implemented/feature/2026-08-20-unified-image-request-pipeline.zh.md)。在 DeepSeek 默认的 128 MiB 预算和 64 MiB 量子下,129 张 1 MiB 的图片会省略最老的 65 张,这个前缀维持到历史超过 192 MiB。没有任何东西记住上一次请求停在哪里,前缀稳定只是因为对 append-only 历史做同样的算术会得到同样的结果。
-
-算术的输入一动,这种稳定就失效。[Files 内联回退](../../implemented/bug-fix/2026-08-21-deepseek-files-inline-fallback.zh.md)会用 20 MiB 内联预算和 10 MiB 量子重建请求,那一次省略多得多的图片,下一次 file 模式的请求又把它们带回来。pi-ai 路由的量子是一个字节,前缀几乎每次请求都会移动。compaction 降低总量,让先前省略的图片回归。切换路由会移动每个台阶边界。每一次移动都改变模型可见前缀,并让 provider 的缓存前缀失效。
-
-同样的重算也破坏了仓库不变量:模型可见输入必须能从 session log 重建。实际发出的表示方式、派生请求版本的精确字节长度、路由预算和量子都是运行时或配置事实,从不进入日志,`request/header` 只记录调用配置、系统提示词和工具。provider usage 只锚定 token 总量,恢复不了图片集合,[按路由定价的估计](../../implemented/feature/2026-08-24-route-priced-image-request-pressure.zh.md)也写明它不复现回退预算。没有任何消费方能把一条已记录的助手响应和它的请求携带的图片集合配对。
-
-## 提案
-
-把 offload 位置变成持久的会话事实:以一条只会前进的、改变表层的会话事件记录图片 offload 水位。
-
-**事件。** 新增一个读取时必须识别的 `SessionEventMap` 成员(工作名 `image/offload`),携带 `{ turn, step }` 和水位位置。它与 compaction 事件同属表层事件:派生历史把位于水位之前的每个图片出现位置标记为已省略,每条路由把被标记的出现位置投影为它的 `offloadedImageText` 占位文本。位于水位及之后的出现位置保留。adapter 保持现有的占位渲染,唯一的变化是省略集合来自表层而不是每次请求的算术。
-
-**只前进。** 预算被超过的路由推进水位,有余量的路由不动它。预算变大、路由切换或 compaction 降低总量时水位永不回退,所以模型可见前缀和 provider 缓存前缀只向前移动。推进保留现有的张数和字节量子,一次推进仍然跳到下一个删除边界,而不是刚好够用的最小值。
-
-**按出现位置定位。** 水位以承载事件的序号加该事件内容中的块路径(含嵌套的工具结果内容)指向第一个保留的图片出现位置。它不用附件 id,因为附件 id 是内容摘要,同一张图附加两次就会重复;也不按出现次数计数,因为 compaction 剪枝会让计数漂移。
-
-**发送前追加。** 与 `request/header` 一样,事件在所属 step 内、请求发出之前追加,因此每个已发出请求的投影在发出时刻已完全由日志决定。file 到内联的回退需要进一步推进时,adapter 在内联发送之前上报需求,循环在同一 step 内追加第二次推进,回退仍然从不发出未记录的投影。
-
-**永不可忽略。** 该事件改变派生表层,不认识该类型的构建必须拒绝这份日志,而不是带着模型从未见过的图片重放它。日志结构没有变化,`SESSION_FORMAT_VERSION` 保持不变。
-
-**决定权在循环。** `LlmRuntime` 根据路由声明的预算(表示方式、字节和张数预算、量子)计算水位并追加事件;adapter 声明预算、上报回退需求,但不决定省略集合。这让 #2644 推迟的路由能力元数据 seam 成为实现的前置条件。
-
-**对其他消费方的影响。** token meter 直接为精确的省略集合定价,不再复现第一轮算术。resume、fork 和重放从日志复现表层。剪掉水位之下消息的 compaction 不影响水位有效性,因为位置指向的是一个保留的出现位置。纯文本路由保留各自的全历史替换,不碰水位。
-
-**范围之外。** 占位和句柄文本中嵌入的执行世界访问路径仍在序列化时解析,这个缺口对保留的图片同样存在,属于另一个关于记录执行世界映射的决定。表层拥有省略集合之后,再用 log-only 事件记录每次请求的投影结果已无必要。
-
-## 考虑过的替代方案
-
-**继续每次请求重算 offload 位置(现状)。** 只在算术输入不动时稳定,内联回退、pi-ai 量子、compaction 和路由切换都会移动前缀,且没有消费方能重建历史请求的图片集合。这正是 issue 要解决的状态。
-
-**用 log-only 事件记录每次请求的投影结果。** 恢复了可重建性但没有稳定性:记录的结果不是决策输入,上述每一种抖动照旧发生,日志只是把它记下来。它还为省略集合制造了两个事实来源,表层算术和记录,两者可能不一致。
-
-**记录完整的投影后请求体。** 除 offload 决定外一切都已可派生,为记录一个位置而每次请求重复整段历史会让日志平方级增长。
-
-**用附件 id 或出现次数标识水位。** 附件 id 在重复附加时重复,位置因此含糊;compaction 剪掉更早的出现位置后计数会漂移。事件序号加块路径没有歧义,也不受剪枝影响。
-
-**让各个 adapter 自己追加事件。** adapter 拥有预算,但不拥有会话表层;在循环之下追加表层事件绕过了循环对派生历史的所有权,也会让两个 adapter 对表层做出不同定义。
-
-## 验收标准
-
-- 水位在 file 模式预算压力、内联回退和切换到更小预算的路由时推进,在 compaction、预算增大或切换回来之后永不回退。
-- 仅凭日志即可确定每个已发出请求的省略集合,包括回退的第二次推进,覆盖 resume、fork、retry 和 compaction。
-- 重放、resume 和 fork 派生出完全相同的表层,不认识该事件类型的构建拒绝这份日志。
-- keyless 录制会话快照覆盖 file 模式省略、内联回退推进和水位之下的 compaction;单元测试钉住位置编码、单调性和发送前的追加时机;两个 SDK 的期望输出为新的表层事件同步更新。
-- token meter 的图片定价消费表层的省略集合,第一轮算术的复现被移除。
-
-## 风险
-
-- 预算变大时被省略的图片不会自动回归,恢复手段是占位文本中的只读路径,模型必须主动使用它。
-- 循环侧决定要求路由通过一个尚不存在的能力元数据 seam 声明预算,没有它实现无法落地。
-- 回退的发送前第二次推进新增了一条 adapter 到循环的通道,不声明范围就会诱使其他临时请求事实进入表层事件。
-- 占位文本中的执行世界路径仍未记录,本提案把重建缺口收窄到这一个输入,而不是完全关闭。

+ 2 - 2
docs/architecture.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write docs/architecture.md
-architecture.md: 902fd7b53fe7493127da68f9a8f381a33b8edc18
-architecture.zh.md: 2890b32b5db1ac30f6eafef47b019a03c6acedd4
+architecture.md: 3ba5a0889bf2f592eec818046d92f713d21c01c7
+architecture.zh.md: a4ab51dc0eb9e1b23f3fa03d3713c545531a66eb

+ 5 - 2
docs/architecture.md

@@ -83,8 +83,9 @@ turn/start
      reject, or a first enter rewritten empty -> close the turn with no step
      step/start
      append entered messages as user/message
+     agent/request -> request/header? -> image/offload?
      derive model history from the log
-     agent/request -> llm/stream -> assistant/chunk* -> assistant/message
+     llm/stream -> assistant/chunk* -> assistant/message
      tool/call* -> tools/pre-execute -> tools/execute -> tools/post-execute -> tool/result*
      step/end
      tools owe another request, or next-step input arrived -> claim -> next step
@@ -92,12 +93,14 @@ turn/start
 turn/end
 ```
 
-`turn/*`, `step/*`, `user/message`, `assistant/*`, and `tool/*` are durable session events; the rest are live extension points across three domains. `agent/pre-step`, `agent/request`, `llm/stream`, and the three `tools/*` events are waterfalls, whose listeners must call `next()` to delegate; `agent/turn-stopping` is serial and has no `next()`.
+`turn/*`, `step/*`, `user/message`, `assistant/*`, `tool/*`, `request/header`, and `image/offload` are durable session events; the rest are live extension points across three domains. `agent/pre-step`, `agent/request`, `llm/stream`, and the three `tools/*` events are waterfalls, whose listeners must call `next()` to delegate; `agent/turn-stopping` is serial and has no `next()`.
 
 Input reaches the driver through one inbox. Some messages wake it immediately; injected context waits in the inbox until another message does.
 
 `agent/pre-step` decides what the model sees. Listeners may rewrite the claimed messages or reject them outright; a rejected or empty first claim still closes a durable turn that spent no step, so the log records the attempt. An enter decision may also set `startsRequestSeries` to begin a distinct model-message series: the loop then logs a fresh `request/header` (reason `series`, or `change` carrying `startsSeries: true` when the envelope changed too). A listener that rebuilds a downstream enter decision must spread it (`{ ...decision, messages }`) so the declaration survives. Each step reads the prompt sections and tool schemas that plugins registered.
 
+`image/offload` records the durable image offload watermark. After `request/header`, when the prepared route declares a request-image budget that the retained image occurrences exceed, the loop appends the advance and only then derives the request, so the history the model receives is exactly the log's projection. An adapter whose exact byte accounting still overflows fails the attempt with `IMAGE_OFFLOAD_REQUIRED` naming the additional oldest occurrences; the loop advances by that count and rebuilds the request before any `agent/request-error` listener runs. The watermark never retreats ([decision](../.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.md)).
+
 Details: the [sequence diagram](agent-lifecycle.md), the [tool pipeline](tool-execution-pipeline.md), and [cancellation and error recovery](subsystems/core.md#the-agent-handle).
 
 ## Session log

+ 5 - 2
docs/architecture.zh.md

@@ -87,8 +87,9 @@ turn/start
      reject, or a first enter rewritten empty -> close the turn with no step
      step/start
      append entered messages as user/message
+     agent/request -> request/header? -> image/offload?
      derive model history from the log
-     agent/request -> llm/stream -> assistant/chunk* -> assistant/message
+     llm/stream -> assistant/chunk* -> assistant/message
      tool/call* -> tools/pre-execute -> tools/execute -> tools/post-execute -> tool/result*
      step/end
      tools owe another request, or next-step input arrived -> claim -> next step
@@ -96,12 +97,14 @@ turn/start
 turn/end
 ```
 
-`turn/*`、`step/*`、`user/message`、`assistant/*` 和 `tool/*` 是持久会话事件;其余是分属三个事件域的实时扩展点。`agent/pre-step`、`agent/request`、`llm/stream` 和三个 `tools/*` 事件是 waterfall(瀑布式事件),其监听器必须调用 `next()` 才能委托下去;`agent/turn-stopping` 是 serial 事件,没有 `next()`。
+`turn/*`、`step/*`、`user/message`、`assistant/*`、`tool/*`、`request/header` 和 `image/offload` 是持久会话事件;其余是分属三个事件域的实时扩展点。`agent/pre-step`、`agent/request`、`llm/stream` 和三个 `tools/*` 事件是 waterfall(瀑布式事件),其监听器必须调用 `next()` 才能委托下去;`agent/turn-stopping` 是 serial 事件,没有 `next()`。
 
 输入通过同一个 inbox 到达驱动器。有些消息会立即唤醒它;注入的上下文会留在 inbox 中,直到另一条消息将其唤醒。
 
 `agent/pre-step` 决定模型看到什么。监听器可以改写已领取的消息,也可以直接拒绝它们;首次领取被拒绝或被改写为空时,仍会关闭一个不含步骤的持久轮次,因此日志会记录这次尝试。enter 决策还可以设置 `startsRequestSeries` 来开启独立的模型消息序列:loop 会随之记录一个新的 `request/header`(原因为 `series`,或在封装同时变化时为携带 `startsSeries: true` 的 `change`)。重建下游 enter 决策的监听器必须展开它(`{ ...decision, messages }`),该声明才能存活。每个步骤读取插件注册的提示词片段和工具 schema。
 
+`image/offload` 记录持久的图片 offload 水位。`request/header` 之后,若已准备的路由声明了请求图片预算且保留的图片出现位置超过该预算,循环先追加推进再派生请求,因此模型收到的历史正是日志的投影。若 adapter 按精确字节计量后仍超预算,会以 `IMAGE_OFFLOAD_REQUIRED` 让本次尝试失败并说明还需省略多少最老的出现位置;循环按该数量推进并在任何 `agent/request-error` 监听器运行之前重建请求。水位永不回退([决定](../.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.zh.md))。
+
 详情见[时序图](agent-lifecycle.zh.md)、[工具流水线](tool-execution-pipeline.zh.md)和[取消与错误恢复](subsystems/core.zh.md#the-agent-handle)。
 
 ## 会话日志

+ 2 - 2
docs/event-producer-consumer.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write docs/event-producer-consumer.md
-event-producer-consumer.md: d04a37ee71756619b36de192011a0053e5ceccb8
-event-producer-consumer.zh.md: 226c368517d989cba2eeac515ee2c544151e8005
+event-producer-consumer.md: 29a80986e2cd1c04f323e6c5ed286b5a5f764e6c
+event-producer-consumer.zh.md: 6e7f1f194e863bf0e723fba43873c181fd077dfe

+ 5 - 5
docs/event-producer-consumer.md

@@ -43,12 +43,12 @@ This matrix shows which packages dispatch each harness-owned event and which pac
 | `fs/write-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:58`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`), [`tool-str-replace-editor`](../packages/fs/tool-str-replace-editor) (`waterfall`) | [`fs-observation-policy`](../packages/fs/fs-observation-policy) |
 | `goal/changed` | `emit` | [`packages/goal/goal/src/domain.ts:114`](../packages/goal/goal/src/domain.ts) | [`goal`](../packages/goal/goal) (`emit`) | [`goal-round-driver`](../packages/goal/goal-round-driver) |
 | `llm/adapters-updated` | `emit` | [`packages/llm/llm/src/types.ts:23`](../packages/llm/llm/src/types.ts) | [`llm`](../packages/llm/llm) (`events.dispatch`) | [`acp`](../packages/acp/acp), [`llm`](../packages/llm/llm), `remotes` |
-| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:67`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/test-support/llm-replay), [`session-checkpoint-policy`](../packages/session/session-checkpoint-policy), [`session-title`](../packages/session/session-title) |
+| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:68`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/test-support/llm-replay), [`session-checkpoint-policy`](../packages/session/session-checkpoint-policy), [`session-title`](../packages/session/session-title) |
 | `session-telemetry/record` | `waterfall` | [`packages/session/session-telemetry/src/index.ts:43`](../packages/session/session-telemetry/src/index.ts) | [`session-telemetry`](../packages/session/session-telemetry) (`waterfall`) | - |
-| `session/created` | `emit` | [`packages/core/session/src/index.ts:52`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compaction`](../packages/compaction/compaction), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm-retry`](../packages/llm/llm-retry), [`permission-presets`](../packages/interaction/permission-presets), [`plan-mode`](../packages/plan/plan-mode), [`schedule`](../packages/schedule/schedule), `server`, [`session`](../packages/core/session), `session-controller`, [`session-log-deepseek`](../packages/session/session-log-deepseek), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title), [`time-context`](../packages/context/time-context), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
-| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:62`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `agent-team`, `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title) |
-| `session/event` | `emit` | [`packages/core/session/src/index.ts:74`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/acp/acp), [`agent-instructions`](../packages/context/agent-instructions), [`agent-loop`](../packages/core/agent-loop), [`agent-presets`](../packages/preset/agent-presets), `agent-team`, [`compaction`](../packages/compaction/compaction), [`compaction-basic`](../packages/compaction/compaction-basic), [`file-reference-local`](../packages/context/file-reference-local), [`goal`](../packages/goal/goal), [`goal-round-driver`](../packages/goal/goal-round-driver), [`headless`](../packages/bundle/headless), [`hook-protocol`](../packages/hooks/hook-protocol), [`loader-smoke`](../packages/test-support/loader-smoke), `server`, [`session`](../packages/core/session), `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-telemetry-otel`](../packages/session/session-telemetry-otel), [`session-title`](../packages/session/session-title), [`token-meter`](../packages/llm/token-meter), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
-| `session/flush` | `parallel` | [`packages/core/session/src/index.ts:83`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-telemetry`](../packages/session/session-telemetry) |
+| `session/created` | `emit` | [`packages/core/session/src/index.ts:54`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compaction`](../packages/compaction/compaction), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm-retry`](../packages/llm/llm-retry), [`permission-presets`](../packages/interaction/permission-presets), [`plan-mode`](../packages/plan/plan-mode), [`schedule`](../packages/schedule/schedule), `server`, [`session`](../packages/core/session), `session-controller`, [`session-log-deepseek`](../packages/session/session-log-deepseek), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title), [`time-context`](../packages/context/time-context), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
+| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:64`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `agent-team`, `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title) |
+| `session/event` | `emit` | [`packages/core/session/src/index.ts:76`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/acp/acp), [`agent-instructions`](../packages/context/agent-instructions), [`agent-loop`](../packages/core/agent-loop), [`agent-presets`](../packages/preset/agent-presets), `agent-team`, [`compaction`](../packages/compaction/compaction), [`compaction-basic`](../packages/compaction/compaction-basic), [`file-reference-local`](../packages/context/file-reference-local), [`goal`](../packages/goal/goal), [`goal-round-driver`](../packages/goal/goal-round-driver), [`headless`](../packages/bundle/headless), [`hook-protocol`](../packages/hooks/hook-protocol), [`loader-smoke`](../packages/test-support/loader-smoke), `server`, [`session`](../packages/core/session), `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-telemetry-otel`](../packages/session/session-telemetry-otel), [`session-title`](../packages/session/session-title), [`token-meter`](../packages/llm/token-meter), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
+| `session/flush` | `parallel` | [`packages/core/session/src/index.ts:85`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-telemetry`](../packages/session/session-telemetry) |
 | `settings/document-updated` | `emit` | [`packages/settings/settings/src/types.ts:105`](../packages/settings/settings/src/types.ts) | [`settings`](../packages/settings/settings) (`events.dispatch`) | `remotes` |
 | `settings/updated` | `emit` | [`packages/settings/settings/src/types.ts:92`](../packages/settings/settings/src/types.ts) | [`settings`](../packages/settings/settings) (`events.dispatch`) | [`settings`](../packages/settings/settings) |
 | `skills/change` | `emit` | [`packages/skill/skill/src/index.ts:298`](../packages/skill/skill/src/index.ts) | [`skill`](../packages/skill/skill) (`events.dispatch`) | - |

+ 5 - 5
docs/event-producer-consumer.zh.md

@@ -45,12 +45,12 @@
 | `fs/write-intent` | `waterfall` | [`packages/fs/fs/src/index.ts:58`](../packages/fs/fs/src/index.ts) | [`tool-fs`](../packages/fs/tool-fs) (`waterfall`), [`tool-str-replace-editor`](../packages/fs/tool-str-replace-editor) (`waterfall`) | [`fs-observation-policy`](../packages/fs/fs-observation-policy) |
 | `goal/changed` | `emit` | [`packages/goal/goal/src/domain.ts:114`](../packages/goal/goal/src/domain.ts) | [`goal`](../packages/goal/goal) (`emit`) | [`goal-round-driver`](../packages/goal/goal-round-driver) |
 | `llm/adapters-updated` | `emit` | [`packages/llm/llm/src/types.ts:23`](../packages/llm/llm/src/types.ts) | [`llm`](../packages/llm/llm) (`events.dispatch`) | [`acp`](../packages/acp/acp), [`llm`](../packages/llm/llm), `remotes` |
-| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:67`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/test-support/llm-replay), [`session-checkpoint-policy`](../packages/session/session-checkpoint-policy), [`session-title`](../packages/session/session-title) |
+| `llm/stream` | `waterfall` | [`packages/llm/llm/src/index.ts:68`](../packages/llm/llm/src/index.ts) | [`llm`](../packages/llm/llm) (`waterfall`) | [`agent-loop`](../packages/core/agent-loop), [`llm`](../packages/llm/llm), [`llm-replay`](../packages/test-support/llm-replay), [`session-checkpoint-policy`](../packages/session/session-checkpoint-policy), [`session-title`](../packages/session/session-title) |
 | `session-telemetry/record` | `waterfall` | [`packages/session/session-telemetry/src/index.ts:43`](../packages/session/session-telemetry/src/index.ts) | [`session-telemetry`](../packages/session/session-telemetry) (`waterfall`) | - |
-| `session/created` | `emit` | [`packages/core/session/src/index.ts:52`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compaction`](../packages/compaction/compaction), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm-retry`](../packages/llm/llm-retry), [`permission-presets`](../packages/interaction/permission-presets), [`plan-mode`](../packages/plan/plan-mode), [`schedule`](../packages/schedule/schedule), `server`, [`session`](../packages/core/session), `session-controller`, [`session-log-deepseek`](../packages/session/session-log-deepseek), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title), [`time-context`](../packages/context/time-context), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
-| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:62`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `agent-team`, `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title) |
-| `session/event` | `emit` | [`packages/core/session/src/index.ts:74`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/acp/acp), [`agent-instructions`](../packages/context/agent-instructions), [`agent-loop`](../packages/core/agent-loop), [`agent-presets`](../packages/preset/agent-presets), `agent-team`, [`compaction`](../packages/compaction/compaction), [`compaction-basic`](../packages/compaction/compaction-basic), [`file-reference-local`](../packages/context/file-reference-local), [`goal`](../packages/goal/goal), [`goal-round-driver`](../packages/goal/goal-round-driver), [`headless`](../packages/bundle/headless), [`hook-protocol`](../packages/hooks/hook-protocol), [`loader-smoke`](../packages/test-support/loader-smoke), `server`, [`session`](../packages/core/session), `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-telemetry-otel`](../packages/session/session-telemetry-otel), [`session-title`](../packages/session/session-title), [`token-meter`](../packages/llm/token-meter), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
-| `session/flush` | `parallel` | [`packages/core/session/src/index.ts:83`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-telemetry`](../packages/session/session-telemetry) |
+| `session/created` | `emit` | [`packages/core/session/src/index.ts:54`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`compaction`](../packages/compaction/compaction), [`goal`](../packages/goal/goal), [`hook-protocol`](../packages/hooks/hook-protocol), [`llm-retry`](../packages/llm/llm-retry), [`permission-presets`](../packages/interaction/permission-presets), [`plan-mode`](../packages/plan/plan-mode), [`schedule`](../packages/schedule/schedule), `server`, [`session`](../packages/core/session), `session-controller`, [`session-log-deepseek`](../packages/session/session-log-deepseek), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title), [`time-context`](../packages/context/time-context), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
+| `session/disposed` | `emit` | [`packages/core/session/src/index.ts:64`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`agent-loop`](../packages/core/agent-loop), `agent-team`, `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-title`](../packages/session/session-title) |
+| `session/event` | `emit` | [`packages/core/session/src/index.ts:76`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`acp`](../packages/acp/acp), [`agent-instructions`](../packages/context/agent-instructions), [`agent-loop`](../packages/core/agent-loop), [`agent-presets`](../packages/preset/agent-presets), `agent-team`, [`compaction`](../packages/compaction/compaction), [`compaction-basic`](../packages/compaction/compaction-basic), [`file-reference-local`](../packages/context/file-reference-local), [`goal`](../packages/goal/goal), [`goal-round-driver`](../packages/goal/goal-round-driver), [`headless`](../packages/bundle/headless), [`hook-protocol`](../packages/hooks/hook-protocol), [`loader-smoke`](../packages/test-support/loader-smoke), `server`, [`session`](../packages/core/session), `session-controller`, [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-projection`](../packages/session/session-projection), [`session-projection-cache`](../packages/session/session-projection-cache), [`session-telemetry`](../packages/session/session-telemetry), [`session-telemetry-otel`](../packages/session/session-telemetry-otel), [`session-title`](../packages/session/session-title), [`token-meter`](../packages/llm/token-meter), [`tool-todo`](../packages/todo/tool-todo), [`tool-workflow`](../packages/workflow/tool-workflow), [`tools`](../packages/core/tools), [`user-approval`](../packages/interaction/user-approval) |
+| `session/flush` | `parallel` | [`packages/core/session/src/index.ts:85`](../packages/core/session/src/index.ts) | [`session`](../packages/core/session) (`events.dispatch`) | [`session-persistence-jsonl`](../packages/session/session-persistence-jsonl), [`session-telemetry`](../packages/session/session-telemetry) |
 | `settings/document-updated` | `emit` | [`packages/settings/settings/src/types.ts:105`](../packages/settings/settings/src/types.ts) | [`settings`](../packages/settings/settings) (`events.dispatch`) | `remotes` |
 | `settings/updated` | `emit` | [`packages/settings/settings/src/types.ts:92`](../packages/settings/settings/src/types.ts) | [`settings`](../packages/settings/settings) (`events.dispatch`) | [`settings`](../packages/settings/settings) |
 | `skills/change` | `emit` | [`packages/skill/skill/src/index.ts:298`](../packages/skill/skill/src/index.ts) | [`skill`](../packages/skill/skill) (`events.dispatch`) | - |

+ 2 - 2
docs/persistence-catalog.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write docs/persistence-catalog.md
-persistence-catalog.md: 1c0c6919987c691b82c4639aff0f779c95dca83c
-persistence-catalog.zh.md: 4cc8ba5b7fc76708a80285013ebcbb03fe3e8e4a
+persistence-catalog.md: ea6c0dbe5be9b4d4332c8de91d19b3f3f90599ef
+persistence-catalog.zh.md: 36da7743c1d404d128ebe36e49303f51291a1b2e

+ 35 - 13
docs/persistence-catalog.md

@@ -90,7 +90,7 @@ export type SessionEvent<T extends SessionEventType = SessionEventType> = {
 }[T]
 ```
 
-Sources: [`packages/core/session/src/types.ts:368`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:375`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:404`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:436`](../packages/core/session/src/types.ts)
+Sources: [`packages/core/session/src/types.ts:393`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:400`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:429`](../packages/core/session/src/types.ts) · [`packages/core/session/src/types.ts:461`](../packages/core/session/src/types.ts)
 
 ## Events
 
@@ -215,7 +215,7 @@ Source: [`packages/interaction/user-approval/src/index.ts:33`](../packages/inter
 
 Types: [StreamChunk](subsystems/llm-streaming.md)
 
-Source: [`packages/core/session/src/types.ts:291`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:305`](../packages/core/session/src/types.ts)
 
 <a id="assistantmessage--surface"></a>
 
@@ -237,7 +237,7 @@ Source: [`packages/core/session/src/types.ts:291`](../packages/core/session/src/
 
 Types: [TokenUsage](subsystems/llm-streaming.md)
 
-Source: [`packages/core/session/src/types.ts:302`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:316`](../packages/core/session/src/types.ts)
 
 ### `command/*`
 
@@ -474,6 +474,28 @@ Source: [`packages/hooks/hook-protocol/src/types.ts:19`](../packages/hooks/hook-
 
 Source: [`packages/hooks/hook-protocol/src/types.ts:31`](../packages/hooks/hook-protocol/src/types.ts)
 
+### `image/*`
+
+<a id="imageoffload--log-only"></a>
+
+#### `image/offload` — log-only
+
+```ts persistence-catalog
+/**
+ * Advances the durable image offload watermark before a request in step
+ * `step` of turn `turn` is dispatched. Every image occurrence positioned at
+ * or before `watermark` derives with `offloaded: true`, so each route sends
+ * its placeholder text instead of the image; occurrences after it stay
+ * retained. The watermark only advances: each event names a position
+ * strictly after the previous one, and no later budget, route change, or
+ * compaction moves it back. It is a log-only event that changes the derived
+ * surface, so a build that does not know the type refuses the log.
+ */
+'image/offload': { turn: number; step: number; watermark: ImageOccurrencePosition }
+```
+
+Source: [`packages/core/session/src/types.ts:366`](../packages/core/session/src/types.ts)
+
 ### `llm/*`
 
 <a id="llmretry--log-only"></a>
@@ -563,7 +585,7 @@ Source: [`packages/plan/plan-mode/src/index.ts:46`](../packages/plan/plan-mode/s
 'request/context': RequestContext
 ```
 
-Source: [`packages/core/session/src/types.ts:341`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:355`](../packages/core/session/src/types.ts)
 
 <a id="requestheader--log-only"></a>
 
@@ -582,7 +604,7 @@ Source: [`packages/core/session/src/types.ts:341`](../packages/core/session/src/
 }
 ```
 
-Source: [`packages/core/session/src/types.ts:331`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:345`](../packages/core/session/src/types.ts)
 
 ### `sandbox/*`
 
@@ -657,7 +679,7 @@ Source: [`packages/schedule/schedule/src/types.ts:219`](../packages/schedule/sch
 'session/end-seed': Record<string, never>
 ```
 
-Source: [`packages/core/session/src/types.ts:364`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:389`](../packages/core/session/src/types.ts)
 
 <a id="sessiontitle--log-only"></a>
 
@@ -717,7 +739,7 @@ Source: [`packages/session/session-log-deepseek/src/types.ts:57`](../packages/se
 'step/end': { turn: number; step: number }
 ```
 
-Source: [`packages/core/session/src/types.ts:281`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:295`](../packages/core/session/src/types.ts)
 
 <a id="stepstart--log-only"></a>
 
@@ -728,7 +750,7 @@ Source: [`packages/core/session/src/types.ts:281`](../packages/core/session/src/
 'step/start': { turn: number; step: number }
 ```
 
-Source: [`packages/core/session/src/types.ts:279`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:293`](../packages/core/session/src/types.ts)
 
 ### `subagent/*`
 
@@ -859,7 +881,7 @@ Source: [`packages/todo/tool-todo/src/types.ts:31`](../packages/todo/tool-todo/s
 
 Types: [ToolCallId](subsystems/core.md)
 
-Source: [`packages/core/session/src/types.ts:308`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:322`](../packages/core/session/src/types.ts)
 
 <a id="toolcode-dispatch--log-only"></a>
 
@@ -934,7 +956,7 @@ Source: [`packages/core/tools/src/types.ts:40`](../packages/core/tools/src/types
 }
 ```
 
-Source: [`packages/core/session/src/types.ts:320`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:334`](../packages/core/session/src/types.ts)
 
 ### `tool-workflow/*`
 
@@ -1014,7 +1036,7 @@ Source: [`packages/workflow/tool-workflow/src/types.ts:47`](../packages/workflow
 
 Types: [TurnEndReason](subsystems/session.md)
 
-Source: [`packages/core/session/src/types.ts:277`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:291`](../packages/core/session/src/types.ts)
 
 <a id="turnstart--log-only"></a>
 
@@ -1030,7 +1052,7 @@ Source: [`packages/core/session/src/types.ts:277`](../packages/core/session/src/
 'turn/start': { turn: number }
 ```
 
-Source: [`packages/core/session/src/types.ts:268`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:282`](../packages/core/session/src/types.ts)
 
 ### `user/*`
 
@@ -1049,7 +1071,7 @@ Source: [`packages/core/session/src/types.ts:268`](../packages/core/session/src/
 'user/message': UserMessage
 ```
 
-Source: [`packages/core/session/src/types.ts:289`](../packages/core/session/src/types.ts)
+Source: [`packages/core/session/src/types.ts:303`](../packages/core/session/src/types.ts)
 
 ### `web/*`
 

+ 22 - 0
docs/persistence-catalog.zh.md

@@ -476,6 +476,28 @@ export type SessionEvent<T extends SessionEventType = SessionEventType> = {
 
 来源:[`packages/hooks/hook-protocol/src/types.ts:31`](../packages/hooks/hook-protocol/src/types.ts)
 
+### `image/*`
+
+<a id="imageoffload--log-only"></a>
+
+#### `image/offload` — log-only
+
+```ts persistence-catalog
+/**
+ * Advances the durable image offload watermark before a request in step
+ * `step` of turn `turn` is dispatched. Every image occurrence positioned at
+ * or before `watermark` derives with `offloaded: true`, so each route sends
+ * its placeholder text instead of the image; occurrences after it stay
+ * retained. The watermark only advances: each event names a position
+ * strictly after the previous one, and no later budget, route change, or
+ * compaction moves it back. It is a log-only event that changes the derived
+ * surface, so a build that does not know the type refuses the log.
+ */
+'image/offload': { turn: number; step: number; watermark: ImageOccurrencePosition }
+```
+
+来源:[`packages/core/session/src/types.ts:366`](../packages/core/session/src/types.ts)
+
 ### `llm/*`
 
 <a id="llmretry--log-only"></a>

+ 2 - 2
docs/subsystems/core.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write docs/subsystems/core.md
-core.md: 170268351454643544d5ab20e3d4b6a8fd8c6c69
-core.zh.md: 048f7ed5f2136bb583c0308635e5cddde2b96785
+core.md: b193b306188e54d4e32985ae2274cfdbc96ea58c
+core.zh.md: 3de620f9bbf48b6affd5797718130baca1de757c

+ 1 - 1
docs/subsystems/core.md

@@ -252,7 +252,7 @@ type SessionStartSource = 'startup' | 'resume' | 'clear' | 'compact'
 
 A `Session` is an **append-only log** of typed `SessionEvent`s — the single source of truth. The LLM message history is *derived* from the log (`deriveMessages()`), not stored separately. Every entry carries a monotonic `seq`, a `time`, and a `type`-discriminated `data` payload; surface variants may also list cited earlier events in `sourceEventSeqs` and carry a `surfaceOp`.
 
-The `SessionEvent` envelope's exact conditional fields, the twelve core event variants (`turn/start`, `turn/end`, `step/start`, `step/end`, `user/message`, `assistant/chunk`, `assistant/message`, `tool/call`, `tool/result`, `request/header`, `request/context`, `session/end-seed`), the `deriveMessages()` projection rules, the `TurnEndReason` reasons, and the execution-enclosure and standalone-event rules are on **[session.md](session.md)**. How the log is made durable — the `SessionPersistence` interface, JSONL provider, `session/flush` checkpoint, crash recovery, and `SessionHeader` — is on **[persistence.md](persistence.md)**.
+The `SessionEvent` envelope's exact conditional fields, the thirteen core event variants (`turn/start`, `turn/end`, `step/start`, `step/end`, `user/message`, `assistant/chunk`, `assistant/message`, `tool/call`, `tool/result`, `request/header`, `request/context`, `image/offload`, `session/end-seed`), the `deriveMessages()` projection rules, the `TurnEndReason` reasons, and the execution-enclosure and standalone-event rules are on **[session.md](session.md)**. How the log is made durable — the `SessionPersistence` interface, JSONL provider, `session/flush` checkpoint, crash recovery, and `SessionHeader` — is on **[persistence.md](persistence.md)**.
 
 ## `ToolDefinition`
 

+ 1 - 1
docs/subsystems/core.zh.md

@@ -260,7 +260,7 @@ type SessionStartSource = 'startup' | 'resume' | 'clear' | 'compact'
 
 `Session` 是一份类型化 `SessionEvent` 的**仅追加日志**——唯一的真源。LLM 消息历史从日志*派生*(`deriveMessages()`),而非单独存储。每个条目携带单调的 `seq`、`time` 与按 `type` 判别的 `data` payload;surface 变体还可以在 `sourceEventSeqs` 中列出被引用的较早事件,并携带 `surfaceOp`。
 
-`SessionEvent` 信封的确切条件字段、十二种核心事件变体(`turn/start`、`turn/end`、`step/start`、`step/end`、`user/message`、`assistant/chunk`、`assistant/message`、`tool/call`、`tool/result`、`request/header`、`request/context`、`session/end-seed`)、`deriveMessages()` 投影规则、`TurnEndReason` 原因以及执行封闭和独立事件规则都在 **[session.md](session.zh.md)** 中。日志如何持久化——`SessionPersistence` 接口、JSONL provider、`session/flush` 检查点、崩溃恢复与 `SessionHeader`——则在 **[persistence.md](persistence.zh.md)** 中。
+`SessionEvent` 信封的确切条件字段、十三种核心事件变体(`turn/start`、`turn/end`、`step/start`、`step/end`、`user/message`、`assistant/chunk`、`assistant/message`、`tool/call`、`tool/result`、`request/header`、`request/context`、`image/offload`、`session/end-seed`)、`deriveMessages()` 投影规则、`TurnEndReason` 原因以及执行封闭和独立事件规则都在 **[session.md](session.zh.md)** 中。日志如何持久化——`SessionPersistence` 接口、JSONL provider、`session/flush` 检查点、崩溃恢复与 `SessionHeader`——则在 **[persistence.md](persistence.zh.md)** 中。
 
 ## `ToolDefinition`
 

+ 2 - 2
docs/subsystems/llm-streaming.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write docs/subsystems/llm-streaming.md
-llm-streaming.md: 6867ae292d77474bcedc1466ae0ce6b1fc1c92d3
-llm-streaming.zh.md: b75e24f2e9010b4fb08c035f14bc4e91dc3971ef
+llm-streaming.md: 86b80be929d0821445b3ad86294e918c89f79b20
+llm-streaming.zh.md: 76e636e1c20824b7b9fdd90c0a7128464df3db3f

+ 15 - 3
docs/subsystems/llm-streaming.md

@@ -233,12 +233,19 @@ interface LlmFailure {
   readonly providerRetryAfterMs?: number
   /** Opaque provider-issued request identifier for diagnostics. */
   readonly requestId?: ProviderRequestId
+  /**
+   * With code `IMAGE_OFFLOAD_REQUIRED`: how many more of the oldest retained
+   * image occurrences the route needs offloaded before the same request fits
+   * its exact byte accounting. The agent loop advances the durable watermark
+   * by this count and rebuilds the request.
+   */
+  readonly offloadImages?: number
 }
 ```
 
 ## Request-image pricing
 
-An adapter whose provider charges visual tokens for request images declares per-route pricing by overriding `LlmAdapter.imageRequestPricing`, and `ctx.llm.imageRequestPricing(provider, model)` resolves it synchronously for consumers. The token meter resolves the routed model's pricing on every measurement so compaction pressure, retention, and range selection price image history as the routed request actually sends it; the DeepSeek adapter reproduces its own request projection (per-model pixel budget, oldest-first offload) and prices retained images with the published v4 vision accounting, while provider usage remains the authoritative anchor for completed requests.
+An adapter whose provider charges visual tokens for request images declares per-route pricing by overriding `LlmAdapter.imageRequestPricing`, and `ctx.llm.imageRequestPricing(provider, model)` resolves it synchronously for consumers. The token meter resolves the routed model's pricing on every measurement so compaction pressure, retention, and range selection price image history as the routed request actually sends it; the DeepSeek adapter prices each retained occurrence at its per-model pixel-budget projection with the published v4 vision accounting and each occurrence the session's `image/offload` watermark marks offloaded as its placeholder text, while provider usage remains the authoritative anchor for completed requests.
 
 ```ts type-equiv
 /**
@@ -267,10 +274,11 @@ interface LlmImageRequestPrice {
 interface LlmImageRequestPricing {
   /**
    * Price every image occurrence of one request projection.
-   * @param images - durable image references in request order, one entry per occurrence.
+   * @param images - surface image blocks in request order, one entry per occurrence; an `offloaded` block
+   *   is priced as its placeholder text.
    * @returns one price per occurrence, aligned by index with `images`.
    */
-  priceImages(images: readonly ImageAttachmentRef[]): readonly LlmImageRequestPrice[]
+  priceImages(images: readonly ImageBlock[]): readonly LlmImageRequestPrice[]
 }
 ```
 
@@ -551,6 +559,8 @@ interface LlmResolvedModelInfo extends LlmModelInfo {
   defaultMaxTokens?: number
   /** Adapter-owned selectable reasoning levels when exposed. */
   reasoning?: LlmModelReasoningInfo
+  /** Request-image budget the route enforces; absent for routes that never offload. */
+  imageRequest?: LlmImageRequestBudget
 }
 ```
 
@@ -739,6 +749,8 @@ interface PreparedLlmCall {
   readonly context?: LlmModelContext
   /** Exact model modalities captured with the adapter dispatch generation. */
   readonly inputModalities?: readonly ModelModality[]
+  /** Detached request-image budget the route enforces, when it declares one. */
+  readonly imageRequest?: LlmImageRequestBudget
   /** Config fields materialized by the captured adapter rather than proposed by the caller. */
   readonly adapterDefaults: LlmCallConfigAdapterDefaults
   /**

+ 15 - 3
docs/subsystems/llm-streaming.zh.md

@@ -235,12 +235,19 @@ interface LlmFailure {
   readonly providerRetryAfterMs?: number
   /** Opaque provider-issued request identifier for diagnostics. */
   readonly requestId?: ProviderRequestId
+  /**
+   * With code `IMAGE_OFFLOAD_REQUIRED`: how many more of the oldest retained
+   * image occurrences the route needs offloaded before the same request fits
+   * its exact byte accounting. The agent loop advances the durable watermark
+   * by this count and rebuilds the request.
+   */
+  readonly offloadImages?: number
 }
 ```
 
 ## 请求图片定价
 
-提供方对请求图片收取视觉 token 的适配器通过覆写 `LlmAdapter.imageRequestPricing` 声明按路由的定价,消费方经 `ctx.llm.imageRequestPricing(provider, model)` 同步解析。token 计量服务在每次计量时解析路由模型的定价,使 compaction 的压力、保留与选段都按路由请求实际发送的形式为图片历史计价;DeepSeek 适配器复现自身的请求投影(按模型的像素预算、最旧优先 offload),并用官方公布的 v4 视觉计量为保留图片定价,已完成请求仍以 provider usage 为权威锚点。
+提供方对请求图片收取视觉 token 的适配器通过覆写 `LlmAdapter.imageRequestPricing` 声明按路由的定价,消费方经 `ctx.llm.imageRequestPricing(provider, model)` 同步解析。token 计量服务在每次计量时解析路由模型的定价,使 compaction 的压力、保留与选段都按路由请求实际发送的形式为图片历史计价;DeepSeek 适配器按模型像素预算的投影用官方公布的 v4 视觉计量为每个保留的出现位置定价,并把会话 `image/offload` 水位标记为已省略的出现位置按其占位文本定价,已完成请求仍以 provider usage 为权威锚点。
 
 ```ts type-equiv
 /**
@@ -269,10 +276,11 @@ interface LlmImageRequestPrice {
 interface LlmImageRequestPricing {
   /**
    * Price every image occurrence of one request projection.
-   * @param images - durable image references in request order, one entry per occurrence.
+   * @param images - surface image blocks in request order, one entry per occurrence; an `offloaded` block
+   *   is priced as its placeholder text.
    * @returns one price per occurrence, aligned by index with `images`.
    */
-  priceImages(images: readonly ImageAttachmentRef[]): readonly LlmImageRequestPrice[]
+  priceImages(images: readonly ImageBlock[]): readonly LlmImageRequestPrice[]
 }
 ```
 
@@ -557,6 +565,8 @@ interface LlmResolvedModelInfo extends LlmModelInfo {
   defaultMaxTokens?: number
   /** Adapter-owned selectable reasoning levels when exposed. */
   reasoning?: LlmModelReasoningInfo
+  /** Request-image budget the route enforces; absent for routes that never offload. */
+  imageRequest?: LlmImageRequestBudget
 }
 ```
 
@@ -745,6 +755,8 @@ interface PreparedLlmCall {
   readonly context?: LlmModelContext
   /** Exact model modalities captured with the adapter dispatch generation. */
   readonly inputModalities?: readonly ModelModality[]
+  /** Detached request-image budget the route enforces, when it declares one. */
+  readonly imageRequest?: LlmImageRequestBudget
   /** Config fields materialized by the captured adapter rather than proposed by the caller. */
   readonly adapterDefaults: LlmCallConfigAdapterDefaults
   /**

+ 2 - 2
docs/subsystems/session.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write docs/subsystems/session.md
-session.md: 48a80ccb92af774db2f5bedea8e2e00043932d31
-session.zh.md: df5cfd631a15bfe015c766842956cc0c344c2ccd
+session.md: 2eaec19f5133003ee49bd2821ce71af9c2112974
+session.zh.md: 3574ebd7fd3cba48cf2fd43f8952545de8f636fb

+ 41 - 1
docs/subsystems/session.md

@@ -105,6 +105,17 @@ interface SessionEventMap {
    * changes. It does not participate in request reconstruction or header equality.
    */
   'request/context': RequestContext
+  /**
+   * Advances the durable image offload watermark before a request in step
+   * `step` of turn `turn` is dispatched. Every image occurrence positioned at
+   * or before `watermark` derives with `offloaded: true`, so each route sends
+   * its placeholder text instead of the image; occurrences after it stay
+   * retained. The watermark only advances: each event names a position
+   * strictly after the previous one, and no later budget, route change, or
+   * compaction moves it back. It is a log-only event that changes the derived
+   * surface, so a build that does not know the type refuses the log.
+   */
+  'image/offload': { turn: number; step: number; watermark: ImageOccurrencePosition }
   /**
    * Marks the end of a constructor seed. Events before it have smaller seq
    * values and came from the seed (resume, fork, or replay); this lifecycle
@@ -175,6 +186,26 @@ interface RequestContext {
 }
 ```
 
+### The image offload event: `image/offload`
+
+The agent loop appends `image/offload` inside a step, after `request/header` and before deriving the request, when the prepared route's request-image budget is exceeded by the surface's retained image occurrences, or when an adapter fails an attempt with `IMAGE_OFFLOAD_REQUIRED`. Its `watermark` names the last offloaded occurrence; `deriveMessages()` marks every occurrence positioned at or before it `offloaded: true`, and each route renders those marks as placeholder text. `Session.append` and seeding reject a watermark that is malformed, names an event outside the log, or does not advance strictly past the previous one, and `session.imageOffloadWatermark()` folds the latest. Like `request/header`, it is not a `SurfaceEventType`; unlike it, it changes the derived surface, so it stays required-on-read ([decision](../../.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.md)).
+
+```ts type-equiv
+/**
+ * Durable position of one image occurrence on the model-visible surface: the
+ * seq of the event carrying it and the block path inside that event's
+ * content (the top-level block index, followed by the index inside a
+ * tool-result block). Positions order by seq, then by path, so a newer
+ * event always lies after an older one regardless of surface replacements.
+ */
+interface ImageOccurrencePosition {
+  /** Seq of the `user/message` or `tool/result` event carrying the occurrence. */
+  seq: SessionSeq
+  /** Block index path inside that event's message content. */
+  path: number[]
+}
+```
+
 ## `SessionEvent<T>` — one log entry
 
 A proper discriminated union over `type` (not independent `type`/`data` unions), so `switch (event.type)` narrows `event.data` without casts. `seq` is the monotonic position in the log (`seq = log.length`); `time` is epoch ms.
@@ -531,6 +562,13 @@ declare class Session {
    * @returns the latest immutable route metadata.
    */
   requestContext(): RequestContext | undefined;
+  /**
+   * The durable image offload watermark in force: every image occurrence
+   * positioned at or before it derives as offloaded. Undefined until the
+   * first `image/offload` event.
+   * @returns the frozen latest watermark, or undefined when nothing is offloaded.
+   */
+  imageOffloadWatermark(): ImageOccurrencePosition | undefined;
   /**
    * Derive the LLM message history by walking the ordered sequences of
    * message-producing events maintained by `surfaceOp` markers. The
@@ -542,7 +580,9 @@ declare class Session {
    *
    * CACHED: each surface node is projected exactly once, when first seen — a
    * call costs O(new nodes), and a surface rewrite (a `replace`;
-   * {@link SessionSurface.replaceGeneration}) rebuilds. The returned array is
+   * {@link SessionSurface.replaceGeneration}) or an `image/offload` advance
+   * rebuilds, and image occurrences at or before the watermark derive with
+   * `offloaded: true` ({@link markImageOffload}). The returned array is
    * a fresh snapshot per call (later appends never grow an array a caller
    * already holds); the `Message` objects in it are SHARED and **deep-frozen**.
    * Their content reuses the already frozen durable event data, so the cache

+ 41 - 1
docs/subsystems/session.zh.md

@@ -105,6 +105,17 @@ interface SessionEventMap {
    * changes. It does not participate in request reconstruction or header equality.
    */
   'request/context': RequestContext
+  /**
+   * Advances the durable image offload watermark before a request in step
+   * `step` of turn `turn` is dispatched. Every image occurrence positioned at
+   * or before `watermark` derives with `offloaded: true`, so each route sends
+   * its placeholder text instead of the image; occurrences after it stay
+   * retained. The watermark only advances: each event names a position
+   * strictly after the previous one, and no later budget, route change, or
+   * compaction moves it back. It is a log-only event that changes the derived
+   * surface, so a build that does not know the type refuses the log.
+   */
+  'image/offload': { turn: number; step: number; watermark: ImageOccurrencePosition }
   /**
    * Marks the end of a constructor seed. Events before it have smaller seq
    * values and came from the seed (resume, fork, or replay); this lifecycle
@@ -175,6 +186,26 @@ interface RequestContext {
 }
 ```
 
+### 图片 offload 事件:`image/offload`
+
+当已准备路由的请求图片预算被表层上保留的图片出现位置超过,或 adapter 以 `IMAGE_OFFLOAD_REQUIRED` 让一次尝试失败时,agent loop 会在 step 内、`request/header` 之后、派生请求之前追加 `image/offload`。其 `watermark` 指向最后一个被省略的出现位置;`deriveMessages()` 把位于它及之前的每个出现位置标为 `offloaded: true`,每条路由把这些标记渲染为占位文本。`Session.append` 与 seed 拒绝畸形、指向日志之外事件或没有严格越过前一条的水位,`session.imageOffloadWatermark()` 折叠最新值。它和 `request/header` 一样不是 `SurfaceEventType`;与之不同的是它改变派生表层,因此读取时必须识别([决定](../../.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.zh.md))。
+
+```ts type-equiv
+/**
+ * Durable position of one image occurrence on the model-visible surface: the
+ * seq of the event carrying it and the block path inside that event's
+ * content (the top-level block index, followed by the index inside a
+ * tool-result block). Positions order by seq, then by path, so a newer
+ * event always lies after an older one regardless of surface replacements.
+ */
+interface ImageOccurrencePosition {
+  /** Seq of the `user/message` or `tool/result` event carrying the occurrence. */
+  seq: SessionSeq
+  /** Block index path inside that event's message content. */
+  path: number[]
+}
+```
+
 ## `SessionEvent<T>`:一条日志条目
 
 基于 `type` 的真正可辨识联合(而非独立的 `type`/`data` 联合),因此 `switch (event.type)` 能直接收窄 `event.data`,无需类型断言。`seq` 是日志中的单调递增位置(`seq = log.length`);`time` 为 epoch 毫秒。
@@ -533,6 +564,13 @@ declare class Session {
    * @returns the latest immutable route metadata.
    */
   requestContext(): RequestContext | undefined;
+  /**
+   * The durable image offload watermark in force: every image occurrence
+   * positioned at or before it derives as offloaded. Undefined until the
+   * first `image/offload` event.
+   * @returns the frozen latest watermark, or undefined when nothing is offloaded.
+   */
+  imageOffloadWatermark(): ImageOccurrencePosition | undefined;
   /**
    * Derive the LLM message history by walking the ordered sequences of
    * message-producing events maintained by `surfaceOp` markers. The
@@ -544,7 +582,9 @@ declare class Session {
    *
    * CACHED: each surface node is projected exactly once, when first seen — a
    * call costs O(new nodes), and a surface rewrite (a `replace`;
-   * {@link SessionSurface.replaceGeneration}) rebuilds. The returned array is
+   * {@link SessionSurface.replaceGeneration}) or an `image/offload` advance
+   * rebuilds, and image occurrences at or before the watermark derive with
+   * `offloaded: true` ({@link markImageOffload}). The returned array is
    * a fresh snapshot per call (later appends never grow an array a caller
    * already holds); the `Message` objects in it are SHARED and **deep-frozen**.
    * Their content reuses the already frozen durable event data, so the cache

+ 2 - 2
packages/core/agent-loop/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/core/agent-loop/README.md
-README.md: 5e4ab0a741f0b7dae821e52f1a9b0a1faa90ceed
-README.zh.md: ef47f897978c8254242075a559f56faa2dceca19
+README.md: 340e05be64f5358fb2b90fe4007ebb29e0d712cc
+README.zh.md: a7fe9d9cde9c2c555fa022ecfdab4ff72dd67690

+ 1 - 1
packages/core/agent-loop/README.md

@@ -88,7 +88,7 @@ The package is the one concrete implementation of the public `Agent` contract. I
 
 ### Request headers and adapter defaults
 
-After `agent/request`, `ctx.llm.prepareCall()` validates adapter-owned fields and resolves reasoning-effort and output-token defaults under the active turn signal. The loop retains that exact adapter through resolution, `request/header` logging, and dispatch. It writes a full header for the first request, a changed envelope, an explicit message-series start, a request after surface replacement, and resume; unchanged steps, retries, and ordinary later turns in the same series inherit the latest header. Before the next waterfall, the loop removes adapter-default fields so the current route resolves them again, while explicit settings persist. An unhandled route still fails with `NO_ADAPTER`.
+After `agent/request`, `ctx.llm.prepareCall()` validates adapter-owned fields and resolves reasoning-effort and output-token defaults under the active turn signal. The loop retains that exact adapter through resolution, `request/header` logging, and dispatch. It writes a full header for the first request, a changed envelope, an explicit message-series start, a request after surface replacement, and resume; unchanged steps, retries, and ordinary later turns in the same series inherit the latest header. Before the next waterfall, the loop removes adapter-default fields so the current route resolves them again, while explicit settings persist. An unhandled route still fails with `NO_ADAPTER`. When the prepared route declares a request-image budget (`PreparedLlmCall.imageRequest`) that the surface's retained image occurrences exceed, the loop appends an `image/offload` advance before deriving the request; an adapter whose exact accounting still overflows fails the attempt with `IMAGE_OFFLOAD_REQUIRED`, and the loop advances the watermark by the named count and rebuilds the request before `agent/request-error` runs.
 
 ### Source map
 

+ 1 - 1
packages/core/agent-loop/README.zh.md

@@ -88,7 +88,7 @@ const handle = await ctx.agents.create({
 
 ### 请求 header 与适配器默认值
 
-`agent/request` 返回后,`ctx.llm.prepareCall()` 会在活跃轮次信号下校验适配器持有的字段,并解析推理强度和输出 token 默认值。循环会在解析、`request/header` 记录与分派期间保留同一个适配器。循环会为首次请求、变化的 envelope、显式消息序列起点、表层替换后的请求及恢复写入完整 header;同一序列内内容未变的步骤、重试与普通后续轮次继承最新 header。下一次 waterfall 前,循环移除适配器默认字段,使当前路由重新解析它们;显式设置则保留。未处理的路由仍以 `NO_ADAPTER` 失败。
+`agent/request` 返回后,`ctx.llm.prepareCall()` 会在活跃轮次信号下校验适配器持有的字段,并解析推理强度和输出 token 默认值。循环会在解析、`request/header` 记录与分派期间保留同一个适配器。循环会为首次请求、变化的 envelope、显式消息序列起点、表层替换后的请求及恢复写入完整 header;同一序列内内容未变的步骤、重试与普通后续轮次继承最新 header。下一次 waterfall 前,循环移除适配器默认字段,使当前路由重新解析它们;显式设置则保留。未处理的路由仍以 `NO_ADAPTER` 失败。当已准备的路由声明了请求图片预算(`PreparedLlmCall.imageRequest`)且表层上保留的图片出现位置超过该预算时,循环会在派生请求前追加一条 `image/offload` 推进;adapter 按精确计量仍超预算时以 `IMAGE_OFFLOAD_REQUIRED` 让本次尝试失败,循环按其说明的数量推进水位并在 `agent/request-error` 运行前重建请求。
 
 ### 源码地图
 

+ 60 - 6
packages/core/agent-loop/src/agent.ts

@@ -16,19 +16,22 @@ import type {
   RequestErrorAction,
 } from '@deepseek-ai/dsh-agent'
 import { Inbox, agentEvents, assembleContextFor } from '@deepseek-ai/dsh-agent'
-import type { GenerateOptions, LlmCallConfig, Message, PreparedLlmCall } from '@deepseek-ai/dsh-llm'
+import type { GenerateOptions, LlmCallConfig, LlmImageRequestBudget, PreparedLlmCall, RetainedImageOccurrence } from '@deepseek-ai/dsh-llm'
 import {
   BlockAssembler,
+  IMAGE_OFFLOAD_REQUIRED_CODE,
   LlmError,
   createAssistantMessage,
   errorChain,
   markAgentLoopRequest,
+  planImageOffload,
+  visitImageBlocks,
 } from '@deepseek-ai/dsh-llm'
 import { deepFreeze } from '@deepseek-ai/dsh-util-values'
 import type { Scope } from '@deepseek-ai/dsh-scope'
 import { createScope } from '@deepseek-ai/dsh-scope'
-import type { EpochHeader, RequestContext, Session, SessionId, SessionSeq, TurnEndReason, UserMessage } from '@deepseek-ai/dsh-session'
-import { canonicalHeader, headerEquals } from '@deepseek-ai/dsh-session'
+import type { EpochHeader, ImageOccurrencePosition, RequestContext, Session, SessionId, SessionSeq, TurnEndReason, UserMessage } from '@deepseek-ai/dsh-session'
+import { canonicalHeader, compareImagePositions, headerEquals } from '@deepseek-ai/dsh-session'
 import { joinContextSections, renderContextSections, renderPrompt } from '@deepseek-ai/dsh-system-prompt'
 import type { PromptAssembly } from '@deepseek-ai/dsh-system-prompt'
 import type {} from '@deepseek-ai/dsh-session-projection'
@@ -352,7 +355,6 @@ export class ReactLoopAgent implements Agent {
         step,
         assembly.tools,
         system,
-        this.session.deriveMessages(),
         startsRequestSeries,
         surfaceGeneration,
         signal,
@@ -388,6 +390,12 @@ export class ReactLoopAgent implements Agent {
         throw error
       }
       const finish = assembler.finish
+      if (finish.kind === 'error'
+        && finish.failure.code === IMAGE_OFFLOAD_REQUIRED_CODE
+        && finish.failure.offloadImages !== undefined
+        && this.advanceImageOffload(turn, step, finish.failure.offloadImages)) {
+        continue
+      }
       if (finish.kind === 'error' || finish.kind === 'aborted') {
         const action = await this.dispatch.waterfall(
           'agent/request-error', {
@@ -446,7 +454,6 @@ export class ReactLoopAgent implements Agent {
     step: number,
     tools: GenerateOptions['tools'] & object,
     system: string,
-    boundaryMessages: Message[],
     startsRequestSeries: boolean,
     surfaceGeneration: number,
     signal: AbortSignal,
@@ -530,11 +537,14 @@ export class ReactLoopAgent implements Agent {
       || previousContext.contextWindow !== requestContext.contextWindow) {
       session.append('request/context', requestContext)
     }
+    if (preparedCall?.imageRequest !== undefined) {
+      this.planImageOffload(turn, step, preparedCall.imageRequest)
+    }
     signal.throwIfAborted()
 
     const request = markAgentLoopRequest(deepFreeze({
       ...header.config,
-      messages: boundaryMessages,
+      messages: session.deriveMessages(),
       ...header.system !== undefined ? { system: header.system } : {},
       ...header.tools !== undefined ? { tools: header.tools } : {},
       sessionId: this.session.id,
@@ -542,4 +552,48 @@ export class ReactLoopAgent implements Agent {
     }))
     return { request, ...preparedCall === undefined ? {} : { preparedCall } }
   }
+
+  /**
+   * Every image occurrence the surface still sends, in log order (oldest
+   * first), with the durable position an `image/offload` advance would name.
+   */
+  private retainedImageOccurrences(): RetainedImageOccurrence<ImageOccurrencePosition>[] {
+    const { session } = this
+    const watermark = session.imageOffloadWatermark()
+    const retained: RetainedImageOccurrence<ImageOccurrencePosition>[] = []
+    for (const seq of session.surface.nodes) {
+      // oxlint-disable-next-line typescript/no-non-null-assertion -- surface nodes index the durable log
+      const message = session.deriveEventMessage(session.eventAt(seq)!)
+      if (message === null) continue
+      visitImageBlocks(message.content, (block, path) => {
+        const position = { seq, path: [...path] }
+        if (watermark !== undefined && compareImagePositions(position, watermark) <= 0) return
+        retained.push({ position, bytes: block.attachment.bytes })
+      })
+    }
+    return retained.sort((a, b) => compareImagePositions(a.position, b.position))
+  }
+
+  /**
+   * Advance the durable watermark when the route budget is exceeded by the
+   * retained occurrences, so the request derived next carries the offloaded
+   * set the log records.
+   */
+  private planImageOffload(turn: number, step: number, budget: LlmImageRequestBudget): void {
+    const watermark = planImageOffload(this.retainedImageOccurrences(), budget)
+    if (watermark !== undefined) this.session.append('image/offload', { turn, step, watermark })
+  }
+
+  /**
+   * Advance the watermark past the `count` oldest retained occurrences an
+   * adapter's exact accounting still cannot send.
+   * @returns whether any occurrence remained to offload, so the step can rebuild the request.
+   */
+  private advanceImageOffload(turn: number, step: number, count: number): boolean {
+    const retained = this.retainedImageOccurrences()
+    const last = retained[Math.min(count, retained.length) - 1]
+    if (last === undefined) return false
+    this.session.append('image/offload', { turn, step, watermark: last.position })
+    return true
+  }
 }

+ 161 - 0
packages/core/agent-loop/tests/image-offload.spec.ts

@@ -0,0 +1,161 @@
+/**
+ * Durable image offload in the loop: the route budget advances the
+ * `image/offload` watermark before dispatch, the watermark never retreats,
+ * and an adapter's `IMAGE_OFFLOAD_REQUIRED` failure advances it by the named
+ * count and rebuilds the request.
+ */
+
+import { describe, expect, it } from 'vitest'
+import { Context } from '@deepseek-ai/cordis'
+import AgentRegistry from '@deepseek-ai/dsh-agent'
+import AgentLoop from '@deepseek-ai/dsh-agent-loop'
+import SessionProjectionRegistry from '@deepseek-ai/dsh-session-projection'
+import LlmRuntime, { createAssistantMessage, createUserMessage, IMAGE_OFFLOAD_REQUIRED_CODE, LlmError, ToolCallId } from '@deepseek-ai/dsh-llm'
+import type { ContentBlock, GenerateOptions, LlmImageRequestBudget } from '@deepseek-ai/dsh-llm'
+import SessionStore, { SessionId } from '@deepseek-ai/dsh-session'
+import type { Session } from '@deepseek-ai/dsh-session'
+import SystemPrompt from '@deepseek-ai/dsh-system-prompt'
+import ToolRuntime from '@deepseek-ai/dsh-tools'
+import { MockAdapter, textResponse } from './mock-adapter.ts'
+
+async function harness(adapter: MockAdapter): Promise<Context> {
+  const ctx = new Context()
+  await ctx.plugin(LlmRuntime)
+  await ctx.plugin(SessionStore)
+  await ctx.plugin(SessionProjectionRegistry)
+  await ctx.plugin(SystemPrompt)
+  await ctx.plugin(ToolRuntime)
+  await ctx.plugin(AgentRegistry)
+  await ctx.plugin(AgentLoop, { agents: [] })
+  ctx.llm.registerAdapter(['mock'], adapter)
+  return ctx
+}
+
+function image(name: string, bytes: number): Extract<ContentBlock, { type: 'image' }> {
+  return {
+    type: 'image',
+    attachment: { attachmentId: `sha256:${'a'.repeat(64)}` as never, name, mediaType: 'image/png', bytes, width: 1, height: 1 },
+  }
+}
+
+function offloadedNames(options: GenerateOptions): string[] {
+  const names: string[] = []
+  for (const message of options.messages) {
+    for (const block of message.content) {
+      if (block.type === 'image' && block.offloaded === true) names.push(block.attachment.name ?? '')
+      if (block.type === 'tool-result') {
+        for (const inner of block.content) {
+          if (inner.type === 'image' && inner.offloaded === true) names.push(inner.attachment.name ?? '')
+        }
+      }
+    }
+  }
+  return names
+}
+
+function offloadEvents(session: Session): { turn: number; step: number; seq: number; path: number[] }[] {
+  return session.snapshotEvents()
+    .filter(event => event.type === 'image/offload')
+    .map(event => ({
+      turn: event.data.turn,
+      step: event.data.step,
+      seq: Number(event.data.watermark.seq),
+      path: event.data.watermark.path,
+    }))
+}
+
+const budget: LlmImageRequestBudget = { representation: 'raw', maxBytes: 10, byteQuantum: 4 }
+
+describe('image/offload in the agent loop', () => {
+  it('advances the watermark before dispatch and never retreats when the budget has headroom again', async () => {
+    const adapter = new MockAdapter([textResponse('one'), textResponse('two')], undefined, undefined, budget)
+    const ctx = await harness(adapter)
+    const agent = await ctx.agentLoop.create(SessionId('offload-advance'), { provider: 'mock', model: 'mock' })
+
+    agent.followup(createUserMessage({
+      content: [image('a', 4), { type: 'tool-result', toolCallId: ToolCallId('shot'), content: [image('b', 4)] }, image('c', 4)],
+      source: { kind: 'user' },
+    }))
+    await agent.whenIdle()
+
+    // 12 raw bytes exceed the 10-byte bound by 2, rounded up to one 4-byte quantum: 'a' alone
+    // removes 4 bytes but the quantum rule crosses into 'b'.
+    expect(offloadedNames(adapter.requests[0]!)).toEqual(['a', 'b'])
+    const events = offloadEvents(agent.session)
+    expect(events).toHaveLength(1)
+    expect(events[0]).toMatchObject({ turn: 1, step: 1, path: [1, 0] })
+    const dispatchOrder = agent.session.snapshotEvents().map(event => event.type)
+    expect(dispatchOrder.indexOf('image/offload')).toBeGreaterThan(dispatchOrder.indexOf('request/header'))
+    expect(dispatchOrder.indexOf('image/offload')).toBeLessThan(dispatchOrder.indexOf('assistant/chunk'))
+
+    // An empty-content assistant node derives no message and carries no occurrence.
+    agent.session.append('assistant/message', {
+      turn: 1,
+      step: 1,
+      message: createAssistantMessage({ content: [], source: { provider: 'mock', model: 'mock' } }),
+    }, { surfaceOp: 'append' })
+    // The next turn adds no bytes beyond the bound, so the watermark holds, and the first
+    // request's offloaded set is still what the second request sends.
+    agent.followup(createUserMessage({ content: [{ type: 'text', text: 'again' }], source: { kind: 'user' } }))
+    await agent.whenIdle()
+    expect(offloadedNames(adapter.requests[1]!)).toEqual(['a', 'b'])
+    expect(offloadEvents(agent.session)).toHaveLength(1)
+  })
+
+  it('leaves a route without a declared budget alone', async () => {
+    const adapter = new MockAdapter([textResponse('ok')])
+    const ctx = await harness(adapter)
+    const agent = await ctx.agentLoop.create(SessionId('offload-none'), { provider: 'mock', model: 'mock' })
+    agent.followup(createUserMessage({ content: [image('a', 400)], source: { kind: 'user' } }))
+    await agent.whenIdle()
+    expect(offloadedNames(adapter.requests[0]!)).toEqual([])
+    expect(offloadEvents(agent.session)).toHaveLength(0)
+  })
+
+  it('advances by the count an IMAGE_OFFLOAD_REQUIRED failure names and rebuilds the request', async () => {
+    const adapter = new MockAdapter([
+      () => {
+        throw new LlmError('inline budget exceeded', IMAGE_OFFLOAD_REQUIRED_CODE, { offloadImages: 2 })
+      },
+      textResponse('sent'),
+    ], undefined, undefined, { representation: 'raw', maxBytes: 100 })
+    const ctx = await harness(adapter)
+    const agent = await ctx.agentLoop.create(SessionId('offload-required'), { provider: 'mock', model: 'mock' })
+    const recoveries: string[] = []
+    ctx.on('agent/request-error', async ({ failure }) => {
+      recoveries.push(failure.code)
+    })
+
+    agent.followup(createUserMessage({
+      content: [image('a', 1), image('b', 1), image('c', 1)], source: { kind: 'user' },
+    }))
+    await agent.whenIdle()
+
+    expect(adapter.requests).toHaveLength(2)
+    expect(offloadedNames(adapter.requests[0]!)).toEqual([])
+    expect(offloadedNames(adapter.requests[1]!)).toEqual(['a', 'b'])
+    expect(offloadEvents(agent.session)).toHaveLength(1)
+    expect(offloadEvents(agent.session)[0]).toMatchObject({ turn: 1, step: 1, path: [1] })
+    expect(recoveries).toEqual([])
+    expect(agent.session.snapshotEvents().filter(event => event.type === 'assistant/message')).toHaveLength(1)
+  })
+
+  it('surfaces IMAGE_OFFLOAD_REQUIRED as an ordinary failure once nothing remains to offload', async () => {
+    const adapter = new MockAdapter([
+      () => {
+        throw new LlmError('still too large', IMAGE_OFFLOAD_REQUIRED_CODE, { offloadImages: 1 })
+      },
+    ])
+    const ctx = await harness(adapter)
+    const agent = await ctx.agentLoop.create(SessionId('offload-exhausted'), { provider: 'mock', model: 'mock' })
+    const recoveries: string[] = []
+    ctx.on('agent/request-error', async ({ failure }) => {
+      recoveries.push(failure.code)
+    })
+    agent.followup(createUserMessage({ content: [{ type: 'text', text: 'no images' }], source: { kind: 'user' } }))
+    await agent.whenIdle()
+    expect(recoveries).toEqual([IMAGE_OFFLOAD_REQUIRED_CODE])
+    expect(offloadEvents(agent.session)).toHaveLength(0)
+    expect(agent.session.snapshotEvents().at(-1)).toMatchObject({ type: 'turn/end', data: { reason: { kind: 'error' } } })
+  })
+})

+ 3 - 1
packages/core/agent-loop/tests/mock-adapter.ts

@@ -1,4 +1,4 @@
-import type { GenerateOptions, LlmModelReasoningInfo, LlmResolvedModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm'
+import type { GenerateOptions, LlmImageRequestBudget, LlmModelReasoningInfo, LlmResolvedModelInfo, StreamChunk } from '@deepseek-ai/dsh-llm'
 import { ToolCallId, LlmAdapter } from '@deepseek-ai/dsh-llm'
 
 /** Helpers to write scripted responses tersely. */
@@ -76,6 +76,7 @@ export class MockAdapter extends LlmAdapter {
     private script: (StreamChunk[] | ((options: GenerateOptions) => StreamChunk[]) | 'hang' | 'hang-slow' | HangAfter)[],
     private readonly reasoning?: LlmModelReasoningInfo,
     private readonly defaultMaxTokens?: number,
+    private readonly imageRequest?: LlmImageRequestBudget,
   ) {
     super()
   }
@@ -90,6 +91,7 @@ export class MockAdapter extends LlmAdapter {
       name: model,
       ...this.reasoning === undefined ? {} : { reasoning: this.reasoning },
       ...this.defaultMaxTokens === undefined ? {} : { defaultMaxTokens: this.defaultMaxTokens },
+      ...this.imageRequest === undefined ? {} : { inputModalities: ['text', 'image'], imageRequest: this.imageRequest },
     })
   }
 

+ 2 - 2
packages/core/session/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/core/session/README.md
-README.md: f5cf910854203021a619cc786dfd13705927ffc1
-README.zh.md: 385d7fd63a6e4dec9c23c9d38a352942d7dbc7f9
+README.md: 810bd9a0620d1491d9e8aeebf656a9ab1a47fc9f
+README.zh.md: 9fd60641c59db28baea14fae523ce3e96bb5b91e

+ 4 - 1
packages/core/session/README.md

@@ -83,6 +83,8 @@ The package is built on event sourcing: a `Session` is an append-only log of typ
 
 `request/header` stores a full canonical snapshot of the non-history request envelope with reason `initial`, `resume`, `change`, or `series`. An explicit message-series start or a surface replacement writes a `series` snapshot when the envelope is unchanged; a simultaneous change uses `startsSeries: true`. Same-series steps, retries, and ordinary later turns inherit the latest snapshot. `adapterDefaults` distinguishes values resolved by the adapter from explicit settings, and `foldRequestHeader()` selects the latest snapshot. This self-contained record supports partial-window rendering and exact reconstruction at the cost of growth per message series; the [reconstructable-requests Agent Note](../../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.md) owns the detail.
 
+`image/offload` records the durable image offload watermark the agent loop advances before dispatch when a route's request-image budget is exceeded: an `ImageOccurrencePosition` naming the last offloaded occurrence by event seq and block path. The derivation marks every occurrence at or before it `offloaded: true`, so each route sends its placeholder text instead of the image; the position only advances, and `Session.append` and seeding reject a malformed or non-advancing watermark. Because the event changes the derived surface it is required-on-read: a build without the type refuses the log rather than replaying images the model never saw ([decision](../../../.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.md)).
+
 ### Source map
 
 | File | Role |
@@ -91,6 +93,7 @@ The package is built on event sourcing: a `Session` is an append-only log of typ
 | [`src/types.ts`](src/types.ts) | `SessionEventMap`, `SessionEvent`, `UserMessage`, `SessionHeader`, `TurnEndReasonMap` |
 | [`src/surface.ts`](src/surface.ts) | Ordered surface projection, replacement validation, `deriveEventMessage` |
 | [`src/request-header.ts`](src/request-header.ts) | `request/header` folding and reconstruction |
+| [`src/image-offload.ts`](src/image-offload.ts) | `image/offload` watermark validation, folding, and surface marking |
 | [`dsh-util-values`](../../util/values/README.md) | Shared lossless JSON validation and detached snapshots |
 | [`src/chunk-rows.ts`](src/chunk-rows.ts) | Shared compact-row storage codec for persistence backends |
 | [`src/repair.ts`](src/repair.ts) | Cold repair of crash-orphaned logs |
@@ -106,7 +109,7 @@ Every append uses the shared iterative `snapshotJsonValue()` pass, which reads,
 
 ### The request header
 
-The loop logs a full canonical `request/header` snapshot (call config, adapter defaults, rendered system prompt, assembled tool schemas) at each loop-instance boundary and on change; `foldRequestHeader(events)` reconstructs it by selecting the latest snapshot, making every conversation request a pure function of the log. Route metadata (`request/context`) is separate logged state appended only when the provider, model, or capacity differs.
+The loop logs a full canonical `request/header` snapshot (call config, adapter defaults, rendered system prompt, assembled tool schemas) at each loop-instance boundary and on change; `foldRequestHeader(events)` reconstructs it by selecting the latest snapshot, making every conversation request a pure function of the log. Route metadata (`request/context`) is separate logged state appended only when the provider, model, or capacity differs. The image offload watermark (`image/offload`) is durable surface state: every image occurrence positioned at or before it derives with `offloaded: true`, each event must advance strictly past the previous one, and `session.imageOffloadWatermark()` folds the latest position.
 
 </details>
 

+ 4 - 1
packages/core/session/README.zh.md

@@ -83,6 +83,8 @@ session.deriveMessages()         // the derived model history
 
 `request/header` 存储非历史请求 envelope 的完整规范快照,原因为 `initial`、`resume`、`change` 或 `series`。显式消息序列起点或表层替换会在 envelope 不变时写入 `series` 快照;同时发生变化时使用 `startsSeries: true`。同一序列内的步骤、重试与普通后续轮次继承最新快照。`adapterDefaults` 区分由适配器解析的值与显式设置,`foldRequestHeader()` 选择最新快照。这种自包含记录以每个消息序列增加存储为代价,支持局部窗口渲染与精确重建;细节由[可重建请求 Agent Note](../../../.agents/notes/implemented/architecture/2026-07-05-reconstructable-requests.zh.md)负责。
 
+`image/offload` 记录 agent loop 在路由请求图片预算被超过时于分派前推进的持久图片 offload 水位:一个以事件序号加块路径指向最后一个被省略出现位置的 `ImageOccurrencePosition`。派生把位于水位及之前的每个出现位置标为 `offloaded: true`,于是每条路由发送其占位文本而不是图片;位置只会推进,`Session.append` 与 seed 都拒绝畸形或不推进的水位。该事件改变派生表层,因此读取时必须识别:不认识该类型的构建拒绝这份日志,而不是带着模型从未见过的图片重放([决定](../../../.agents/notes/implemented/architecture/2026-09-02-image-offload-watermark.zh.md))。
+
 ### 源码地图
 
 | 文件 | 职责 |
@@ -91,6 +93,7 @@ session.deriveMessages()         // the derived model history
 | [`src/types.ts`](src/types.ts) | `SessionEventMap`、`SessionEvent`、`UserMessage`、`SessionHeader`、`TurnEndReasonMap` |
 | [`src/surface.ts`](src/surface.ts) | 有序 surface 投影、替换校验、`deriveEventMessage` |
 | [`src/request-header.ts`](src/request-header.ts) | `request/header` 折叠与重建 |
+| [`src/image-offload.ts`](src/image-offload.ts) | `image/offload` 水位校验、折叠与表层标记 |
 | [`dsh-util-values`](../../util/values/README.zh.md) | 共享无损 JSON 校验与分离式快照 |
 | [`src/chunk-rows.ts`](src/chunk-rows.ts) | 供持久化后端使用的共享紧凑行存储编解码器 |
 | [`src/repair.ts`](src/repair.ts) | 崩溃遗留日志的冷修复 |
@@ -106,7 +109,7 @@ session.deriveMessages()         // the derived model history
 
 ### 请求头
 
-循环在每个循环实例边界及变更时记录完整规范 `request/header` 快照(调用配置、适配器默认值、渲染后的系统提示词、组装后的工具 schema);`foldRequestHeader(events)` 通过选择最新快照来重建它,使每个对话请求都成为日志的纯函数。路由元数据(`request/context`)是独立的已记录状态,仅在提供方、模型或容量变化时追加。
+循环在每个循环实例边界及变更时记录完整规范 `request/header` 快照(调用配置、适配器默认值、渲染后的系统提示词、组装后的工具 schema);`foldRequestHeader(events)` 通过选择最新快照来重建它,使每个对话请求都成为日志的纯函数。路由元数据(`request/context`)是独立的已记录状态,仅在提供方、模型或容量变化时追加。图片 offload 水位(`image/offload`)是持久的表层状态:位于水位及之前的每个图片出现位置派生时带 `offloaded: true`,每条事件必须严格越过前一条,`session.imageOffloadWatermark()` 折叠最新位置。
 
 </details>
 

+ 98 - 0
packages/core/session/src/image-offload.ts

@@ -0,0 +1,98 @@
+/**
+ * Durable image offload watermark: the `image/offload` fold, its append-time
+ * validation, and the per-node projection that marks image occurrences at or
+ * before the watermark as offloaded.
+ *
+ * @module @deepseek-ai/dsh-session/image-offload
+ */
+
+import { compareImageBlockPaths, markOffloadedImages } from '@deepseek-ai/dsh-llm'
+import type { Message } from '@deepseek-ai/dsh-llm'
+import type { ImageOccurrencePosition, SessionEvent, SessionSeq } from './types.ts'
+
+/**
+ * Compare two image occurrence positions in log order: by event seq, then by
+ * block path inside the event's content.
+ * @param a - first position.
+ * @param b - second position.
+ * @returns negative, zero, or positive as `a` lies before, at, or after `b`.
+ */
+export function compareImagePositions(a: ImageOccurrencePosition, b: ImageOccurrencePosition): number {
+  return a.seq === b.seq ? compareImageBlockPaths(a.path, b.path) : a.seq - b.seq
+}
+
+/**
+ * Select the watermark in force after a run of events: the last
+ * `image/offload` event's position, or the prior watermark when the run
+ * carries none.
+ * @param events - events in seq order.
+ * @param prior - watermark folded from the events before `events`.
+ * @returns the latest watermark, or undefined when no event has set one.
+ */
+export function foldImageOffloadWatermark(
+  events: readonly SessionEvent[],
+  prior?: ImageOccurrencePosition,
+): ImageOccurrencePosition | undefined {
+  let watermark = prior
+  for (const event of events) {
+    if (event.type === 'image/offload') watermark = event.data.watermark
+  }
+  return watermark
+}
+
+/**
+ * Validate one `image/offload` payload at the append or seed boundary: the
+ * position must name an event already in the log, carry a block path of
+ * non-negative integers, and lie strictly after the watermark in force.
+ * @param data - the event payload as materialized JSON.
+ * @param logLength - number of events already accepted.
+ * @param prior - watermark in force before this event.
+ * @param location - append or seed location named by the diagnostic.
+ * @throws when the payload is malformed or does not advance the watermark.
+ */
+export function assertImageOffloadAdvance(
+  data: unknown,
+  logLength: number,
+  prior: ImageOccurrencePosition | undefined,
+  location: string,
+): void {
+  const record = data !== null && typeof data === 'object' && !Array.isArray(data)
+    ? data as Record<string, unknown>
+    : undefined
+  const watermark = record?.['watermark']
+  const position = watermark !== null && typeof watermark === 'object' && !Array.isArray(watermark)
+    ? watermark as Record<string, unknown>
+    : undefined
+  const seq = position?.['seq']
+  const path = position?.['path']
+  if (typeof seq !== 'number' || !Number.isSafeInteger(seq) || seq < 0 || seq >= logLength
+    || !Array.isArray(path) || path.length === 0
+    || path.some(index => typeof index !== 'number' || !Number.isSafeInteger(index) || index < 0)) {
+    throw new Error(`${location} names an invalid image offload watermark`)
+  }
+  if (prior !== undefined
+    && compareImagePositions({ seq: seq as SessionSeq, path: path as number[] }, prior) <= 0) {
+    throw new Error(`${location} does not advance the image offload watermark`)
+  }
+}
+
+/**
+ * Project one derived surface message under the watermark: every image
+ * occurrence positioned at or before it carries `offloaded: true`.
+ * @param message - the node's derived message.
+ * @param seq - the node's event seq.
+ * @param watermark - watermark in force, or undefined when nothing is offloaded.
+ * @returns the same message when nothing changes, otherwise a shallow copy with marked content.
+ */
+export function markImageOffload(
+  message: Message,
+  seq: SessionSeq,
+  watermark: ImageOccurrencePosition | undefined,
+): Message {
+  if (watermark === undefined || seq > watermark.seq) return message
+  const content = markOffloadedImages(
+    message.content,
+    path => seq < watermark.seq || compareImageBlockPaths(path, watermark.path) <= 0,
+  )
+  return content === message.content ? message : { ...message, content: content as Message['content'] }
+}

+ 43 - 4
packages/core/session/src/index.ts

@@ -15,10 +15,11 @@ import type { Scoped } from '@deepseek-ai/dsh-scope'
 import type { Message } from '@deepseek-ai/dsh-llm'
 import { SESSION_FORMAT_VERSION, SessionLogOffset, SessionSeq } from './types.ts'
 import type { TypertLookup } from '@deepseek-ai/dsh-typert-protocol'
-import type { CreateSessionOptions, EpochHeader, PrepareSessionOptions, RequestContext, SessionEvent, SessionEventMap, SessionEventType, SessionHeader, SessionId, SurfaceIntent, SurfaceEventType } from './types.ts'
+import type { CreateSessionOptions, EpochHeader, ImageOccurrencePosition, PrepareSessionOptions, RequestContext, SessionEvent, SessionEventMap, SessionEventType, SessionHeader, SessionId, SurfaceIntent, SurfaceEventType } from './types.ts'
 import { deriveEventMessage, SurfaceManager } from './surface.ts'
 import type { SessionSurface } from './surface.ts'
 import { foldRequestHeader } from './request-header.ts'
+import { assertImageOffloadAdvance, markImageOffload } from './image-offload.ts'
 
 export * from './types.ts'
 export { SessionPreparation } from './preparation.ts'
@@ -30,6 +31,7 @@ export type { ChunkRow, StorageRecord } from './chunk-rows.ts'
 export type { SessionSurface, SurfaceFoldReplacement, SurfaceFoldResult } from './surface.ts'
 export { deriveEventMessage, foldSurface, isAppendSurfaceEvent, isReplacementSurfaceEvent, isSurfaceEvent, isSurfaceEligibleType } from './surface.ts'
 export { canonicalHeader, foldRequestHeader, headerEquals } from './request-header.ts'
+export { assertImageOffloadAdvance, compareImagePositions, foldImageOffloadWatermark, markImageOffload } from './image-offload.ts'
 export { KNOWN_SESSION_EVENT_TYPES } from './known-event-types.ts'
 
 declare module '@deepseek-ai/cordis' {
@@ -424,6 +426,8 @@ const attachments = new WeakMap<Session, SessionEntry>()
  */
 export class Session {
   private log: SessionEvent[] = []
+  /** Watermark in force after the last accepted `image/offload` event. */
+  private imageOffloadFold: ImageOccurrencePosition | undefined
   /** Single incremental owner of surface acceptance and projection state. */
   private readonly surfaceManager = new SurfaceManager(this.log)
 
@@ -538,6 +542,9 @@ export class Session {
         }
         assertSessionEventEnvelope(snapshot, index)
         assertSupportedRequestHeader(snapshot.type, snapshot.data, `seed event at index ${index}`)
+        if (snapshot.type === 'image/offload') {
+          assertImageOffloadAdvance(snapshot.data, this.log.length, this.imageOffloadFold, `seed event at index ${index}`)
+        }
         if (snapshot.seq !== index) {
           throw new Error(`seed event at index ${index} has seq ${snapshot.seq} (expected ${index}); seed must be contiguous from 0`)
         }
@@ -550,6 +557,7 @@ export class Session {
           throw new Error(`invalid seed event at index ${index}: ${error instanceof Error ? error.message : 'invalid surface metadata'}`)
         }
         this.log.push(mode === 'restore' ? freezeRestoredObject(snapshot) : deepFreeze(snapshot))
+        this.foldImageOffload(snapshot)
       }
     }
     this.firstLiveSeq = SessionLogOffset(this.log.length)
@@ -680,6 +688,9 @@ export class Session {
       throw new Error(`session event "${type}" carries non-JSON-serializable data`)
     }
     assertSupportedRequestHeader(type, dataSnapshot, `session event "${type}"`)
+    if (type === 'image/offload') {
+      assertImageOffloadAdvance(dataSnapshot, this.log.length, this.imageOffloadFold, `session event "${type}"`)
+    }
     const surfaceMetadataSnapshot = snapshotJsonValue(surfaceMetadata)
     if (surfaceMetadataSnapshot === undefined) {
       throw new Error(`session event "${type}" carries non-JSON-serializable surface metadata`)
@@ -705,6 +716,7 @@ export class Session {
         callbacks = collectSessionCallbacks(entry.emitCtx, [entry.carrier, 'session/event', ...callbackArgs])
       }
       this.log.push(event as SessionEvent)
+      this.foldImageOffload(event as SessionEvent)
       this.eventsSnapshot = undefined
       if (callbacks !== undefined && entry !== undefined) {
         invokeContainedSessionObservers(entry.emitCtx, 'session/event', entry.id, callbackArgs, callbacks)
@@ -762,12 +774,29 @@ export class Session {
     return this.contextFold
   }
 
+  /** Record an accepted `image/offload` event as the watermark in force. */
+  private foldImageOffload(event: SessionEvent): void {
+    if (event.type === 'image/offload') this.imageOffloadFold = event.data.watermark
+  }
+
+  /**
+   * The durable image offload watermark in force: every image occurrence
+   * positioned at or before it derives as offloaded. Undefined until the
+   * first `image/offload` event.
+   * @returns the frozen latest watermark, or undefined when nothing is offloaded.
+   */
+  imageOffloadWatermark(): ImageOccurrencePosition | undefined {
+    return this.imageOffloadFold
+  }
+
   /** The derived-message cache: frozen projections, extended per unseen node. */
   private derived: Message[] = []
   /** Surface position (nodes projected) the cache has reached. */
   private derivedNodes = 0
   /** {@link SurfaceManager.replaceGeneration} the cache was built under. */
   private derivedGeneration = 0
+  /** Watermark the cache was built under; an advance rebuilds every node. */
+  private derivedWatermark: ImageOccurrencePosition | undefined
 
   /**
    * Derive the LLM message history by walking the ordered sequences of
@@ -780,7 +809,9 @@ export class Session {
    *
    * CACHED: each surface node is projected exactly once, when first seen — a
    * call costs O(new nodes), and a surface rewrite (a `replace`;
-   * {@link SessionSurface.replaceGeneration}) rebuilds. The returned array is
+   * {@link SessionSurface.replaceGeneration}) or an `image/offload` advance
+   * rebuilds, and image occurrences at or before the watermark derive with
+   * `offloaded: true` ({@link markImageOffload}). The returned array is
    * a fresh snapshot per call (later appends never grow an array a caller
    * already holds); the `Message` objects in it are SHARED and **deep-frozen**.
    * Their content reuses the already frozen durable event data, so the cache
@@ -791,10 +822,12 @@ export class Session {
     const surface = this.surface
     const nodes = surface.nodes
     const generation = surface.replaceGeneration
-    if (generation !== this.derivedGeneration) {
+    const watermark = this.imageOffloadFold
+    if (generation !== this.derivedGeneration || watermark !== this.derivedWatermark) {
       this.derived = []
       this.derivedNodes = 0
       this.derivedGeneration = generation
+      this.derivedWatermark = watermark
     }
     for (const seq of nodes.slice(this.derivedNodes)) {
       // Surface sequences are built from this.log — seq is always a valid
@@ -804,12 +837,18 @@ export class Session {
       // A surface node is one of the five message-producing types, but an
       // empty-content assistant/message (a max-tokens step that hosts only
       // usage) derives to null and must not enter the transcript.
-      if (msg) this.derived.push(msg)
+      if (msg) this.derived.push(this.markImageOffload(msg, seq, watermark))
     }
     this.derivedNodes = nodes.length
     return [...this.derived]
   }
 
+  /** Apply the watermark to one derived node, freezing any marked copy like the durable original. */
+  private markImageOffload(message: Message, seq: SessionSeq, watermark: ImageOccurrencePosition | undefined): Message {
+    const marked = markImageOffload(message, seq, watermark)
+    return marked === message ? message : deepFreeze(marked)
+  }
+
   /**
    * Instance face of the pure per-node `deriveEventMessage` export from
    * `surface.ts`.

+ 2 - 1
packages/core/session/src/invariant.ts

@@ -148,7 +148,8 @@ function validateEvent(
       // Unconstrained: an unbalanced seed legally puts it inside an open turn.
       break
     case 'request/header':
-    case 'request/context': {
+    case 'request/context':
+    case 'image/offload': {
       if (trace.openTurn === null) {
         fail(`${event.type} appended outside any open turn (core execution events must be turn-enclosed)`)
       }

+ 1 - 0
packages/core/session/src/known-event-types.ts

@@ -37,6 +37,7 @@ export const KNOWN_SESSION_EVENT_TYPES: ReadonlySet<string> = new Set([
   'goal/change',
   'hook/invoked',
   'hook/result',
+  'image/offload',
   'llm/retry',
   'llm/retry-started',
   'model/selection',

+ 25 - 0
packages/core/session/src/types.ts

@@ -242,6 +242,20 @@ export interface RequestContext {
   contextWindow?: number
 }
 
+/**
+ * Durable position of one image occurrence on the model-visible surface: the
+ * seq of the event carrying it and the block path inside that event's
+ * content (the top-level block index, followed by the index inside a
+ * tool-result block). Positions order by seq, then by path, so a newer
+ * event always lies after an older one regardless of surface replacements.
+ */
+export interface ImageOccurrencePosition {
+  /** Seq of the `user/message` or `tool/result` event carrying the occurrence. */
+  seq: SessionSeq
+  /** Block index path inside that event's message content. */
+  path: number[]
+}
+
 /**
  * Why a `request/header` snapshot was appended: `'initial'` — the log's first
  * header (a new conversation); `'resume'` — a loop instance's first request
@@ -339,6 +353,17 @@ export interface SessionEventMap {
    * changes. It does not participate in request reconstruction or header equality.
    */
   'request/context': RequestContext
+  /**
+   * Advances the durable image offload watermark before a request in step
+   * `step` of turn `turn` is dispatched. Every image occurrence positioned at
+   * or before `watermark` derives with `offloaded: true`, so each route sends
+   * its placeholder text instead of the image; occurrences after it stay
+   * retained. The watermark only advances: each event names a position
+   * strictly after the previous one, and no later budget, route change, or
+   * compaction moves it back. It is a log-only event that changes the derived
+   * surface, so a build that does not know the type refuses the log.
+   */
+  'image/offload': { turn: number; step: number; watermark: ImageOccurrencePosition }
   /**
    * Marks the end of a constructor seed. Events before it have smaller seq
    * values and came from the seed (resume, fork, or replay); this lifecycle

+ 161 - 0
packages/core/session/tests/image-offload.spec.ts

@@ -0,0 +1,161 @@
+/**
+ * Durable image offload watermark: append/seed validation, monotonic
+ * advance, derived-surface marking, cache rebuild, and replay equality.
+ */
+
+import { describe, expect, it } from 'vitest'
+import { createUserMessage, ToolCallId } from '@deepseek-ai/dsh-llm'
+import type { ContentBlock } from '@deepseek-ai/dsh-llm'
+import { compareImagePositions, foldImageOffloadWatermark, Session, SessionId, SessionSeq } from '@deepseek-ai/dsh-session'
+
+function image(name: string): Extract<ContentBlock, { type: 'image' }> {
+  return {
+    type: 'image',
+    attachment: {
+      attachmentId: `sha256:${'a'.repeat(64)}` as never,
+      name,
+      mediaType: 'image/png',
+      bytes: 3,
+      width: 1,
+      height: 1,
+    },
+  }
+}
+
+function toolResult(name: string): ContentBlock {
+  return { type: 'tool-result', toolCallId: ToolCallId(name), content: [{ type: 'text', text: 'shot' }, image(name)] }
+}
+
+function seeded(): Session {
+  const session = Session.create(SessionId('offload'))
+  session.append('turn/start', { turn: 1 })
+  session.append('step/start', { turn: 1, step: 1 })
+  session.append('user/message', createUserMessage({
+    content: [{ type: 'text', text: 'no images here' }], source: { kind: 'user' },
+  }), { surfaceOp: 'append' })
+  session.append('user/message', createUserMessage({
+    content: [image('first'), image('second')], source: { kind: 'user' },
+  }), { surfaceOp: 'append' })
+  session.append('user/message', createUserMessage({
+    content: [toolResult('third'), image('fourth')], source: { kind: 'user' },
+  }), { surfaceOp: 'append' })
+  return session
+}
+
+function offloadedNames(session: Session): string[] {
+  const names: string[] = []
+  for (const message of session.deriveMessages()) {
+    for (const block of message.content) {
+      if (block.type === 'image' && block.offloaded === true) names.push(block.attachment.name ?? '')
+      if (block.type === 'tool-result') {
+        for (const inner of block.content) {
+          if (inner.type === 'image' && inner.offloaded === true) names.push(inner.attachment.name ?? '')
+        }
+      }
+    }
+  }
+  return names
+}
+
+describe('image/offload append validation', () => {
+  it('rejects a malformed or out-of-range watermark', () => {
+    const session = seeded()
+    expect(() => session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(99), path: [0] } }))
+      .toThrow('names an invalid image offload watermark')
+    expect(() => session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [] } }))
+      .toThrow('names an invalid image offload watermark')
+    expect(() => session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [-1] } }))
+      .toThrow('names an invalid image offload watermark')
+    expect(() => session.append('image/offload', { turn: 1, step: 1, watermark: null as never }))
+      .toThrow('names an invalid image offload watermark')
+    expect(() => session.append('image/offload', 7 as never))
+      .toThrow('names an invalid image offload watermark')
+    expect(session.imageOffloadWatermark()).toBeUndefined()
+  })
+
+  it('accepts only strictly advancing positions', () => {
+    const session = seeded()
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [1] } })
+    expect(session.imageOffloadWatermark()).toEqual({ seq: 3, path: [1] })
+    expect(() => session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [1] } }))
+      .toThrow('does not advance the image offload watermark')
+    expect(() => session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [0] } }))
+      .toThrow('does not advance the image offload watermark')
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(4), path: [0, 1] } })
+    expect(session.imageOffloadWatermark()).toEqual({ seq: 4, path: [0, 1] })
+    expect(Object.isFrozen(session.imageOffloadWatermark())).toBe(true)
+  })
+
+  it('validates a seed with the same rules and folds its watermark', () => {
+    const session = seeded()
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [0] } })
+    const replayed = Session.create(SessionId('offload-replay'), session.snapshotEvents())
+    expect(replayed.imageOffloadWatermark()).toEqual({ seq: 3, path: [0] })
+    expect(replayed.deriveMessages()).toEqual(session.deriveMessages())
+
+    const events = session.snapshotEvents().map(event => (
+      event.type === 'image/offload'
+        ? { ...event, data: { ...event.data, watermark: { seq: 50, path: [0] } } }
+        : event
+    ))
+    expect(() => Session.create(SessionId('offload-bad-seed'), events as never))
+      .toThrow('seed event at index 5 names an invalid image offload watermark')
+  })
+})
+
+describe('image/offload surface derivation', () => {
+  it('marks occurrences at or before the watermark, including nested tool results', () => {
+    const session = seeded()
+    expect(offloadedNames(session)).toEqual([])
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [0] } })
+    expect(offloadedNames(session)).toEqual(['first'])
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(4), path: [0, 1] } })
+    expect(offloadedNames(session)).toEqual(['first', 'second', 'third'])
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(4), path: [1] } })
+    expect(offloadedNames(session)).toEqual(['first', 'second', 'third', 'fourth'])
+  })
+
+  it('freezes marked copies, keeps the durable event content untouched, and matches a scratch replay', () => {
+    const session = seeded()
+    const before = session.deriveMessages()
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [0] } })
+    const after = session.deriveMessages()
+    expect(after[0]).toBe(before[0])
+    expect(after[1]).not.toBe(before[1])
+    expect(Object.isFrozen(after[1])).toBe(true)
+    expect(Object.isFrozen(after[1]!.content[0])).toBe(true)
+    expect(after[2]).toBe(before[2])
+    expect(before[1]!.content[0]).toEqual(image('first'))
+    expect(session.eventAt(SessionSeq(3))!.data).toMatchObject({ content: [image('first'), image('second')] })
+    const scratch = Session.create(SessionId('offload-scratch'), session.snapshotEvents())
+    expect(scratch.deriveMessages()).toEqual(after)
+  })
+
+  it('leaves later appended nodes retained without rebuilding the cache', () => {
+    const session = seeded()
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(4), path: [1] } })
+    const marked = session.deriveMessages()
+    session.append('user/message', createUserMessage({
+      content: [image('fifth')], source: { kind: 'user' },
+    }), { surfaceOp: 'append' })
+    const grown = session.deriveMessages()
+    expect(grown[0]).toBe(marked[0])
+    expect(grown.at(-1)!.content[0]).toEqual(image('fifth'))
+    expect(offloadedNames(session)).toEqual(['first', 'second', 'third', 'fourth'])
+  })
+})
+
+describe('image offload helpers', () => {
+  it('orders positions by seq then path and folds the latest watermark', () => {
+    expect(compareImagePositions({ seq: SessionSeq(1), path: [5] }, { seq: SessionSeq(3), path: [0] })).toBeLessThan(0)
+    expect(compareImagePositions({ seq: SessionSeq(4), path: [1] }, { seq: SessionSeq(3), path: [0, 3] })).toBeGreaterThan(0)
+    const session = seeded()
+    expect(foldImageOffloadWatermark(session.snapshotEvents())).toBeUndefined()
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(3), path: [0] } })
+    const prior = foldImageOffloadWatermark(session.snapshotEvents())
+    expect(prior).toEqual({ seq: 3, path: [0] })
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: SessionSeq(4), path: [1] } })
+    expect(foldImageOffloadWatermark(session.snapshotEvents().slice(6), prior)).toEqual({ seq: 4, path: [1] })
+    expect(foldImageOffloadWatermark([], prior)).toEqual(prior)
+  })
+})

+ 2 - 2
packages/llm/llm-deepseek/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/llm/llm-deepseek/README.md
-README.md: 6c69083909c71631dde04b453b399cc6bf687110
-README.zh.md: 9b6d35864dee20320e3f16bed82d8eecb4f39ec1
+README.md: da64841f610a9099360eee878e7fff8b35ed29bd
+README.zh.md: 6863a991b6d86460e673f39f83980675d07e0a18

+ 2 - 2
packages/llm/llm-deepseek/README.md

@@ -58,7 +58,7 @@ A request selects the route with `provider: deepseek-official`; the model id pas
 | `defaultContextWindow` | `1,000,000` | Capacity fallback for models without an exact value |
 | `models` | V4 Flash + V4 Pro + V4 Flash Vision Exp | Advisory catalog shown by discovery consumers |
 | `streamIdleTimeoutMs` | `300,000` | Maximum provider idle time per outstanding stream read |
-| `maxRequestFilesBytes` | `128 MiB` | High watermark for retained request-image bytes before oldest-first offload |
+| `maxRequestFilesBytes` | `128 MiB` | File-mode request-image byte budget declared to the agent loop's `image/offload` watermark |
 | `maxInlineRequestImageBytes` | `20 MiB` | Independent base64 fallback high watermark |
 | `maxImagesPerRequest` | `600` | High watermark for retained request-image count |
 | `imageOffloadByteQuantum` | `64 MiB` | Files-mode oldest-prefix removal quantum |
@@ -156,7 +156,7 @@ The selected DeepSeek model receives the harness system prompt, message history,
 
 #### Token effect
 
-Provider tokenization governs exact text and image-token input. The adapter declares per-route `imageRequestPricing`: it reproduces oldest-first image offload from durable byte lengths and prices each retained image at its projected dimensions with the published v4 vision accounting (14px patch grid, 3:1 downsampling, 384-token cap, worst-case alignment pad). This lets the token meter price image pressure before a request; reported usage remains authoritative. Reasoning passback carries every reasoned turn's chain of thought into later requests, while dropping over-budget images avoids paying those tokens again. Cache-read usage is reported when available. `totalTokens` is the exact `prompt_tokens + completion_tokens` aggregate and is omitted if a supplied `total_tokens` disagrees.
+Provider tokenization governs exact text and image-token input. The adapter declares per-route `imageRequestPricing`: it prices each occurrence the session's `image/offload` watermark marks offloaded as its placeholder text and each retained image at its projected dimensions with the published v4 vision accounting (14px patch grid, 3:1 downsampling, 384-token cap, worst-case alignment pad). This lets the token meter price image pressure before a request; reported usage remains authoritative. Reasoning passback carries every reasoned turn's chain of thought into later requests, while offloaded images stop costing visual tokens. Each image-capable catalog route also declares its file-mode budget (`maxRequestFilesBytes`, `maxImagesPerRequest`, both quanta, and the request-version byte target) as `imageRequest` on its resolved model info, so the agent loop advances the watermark from logged facts; a request whose retained occurrences still exceed the file-mode or inline-fallback budget at their exact request-version bytes fails with `IMAGE_OFFLOAD_REQUIRED` naming the additional oldest occurrences to offload. Cache-read usage is reported when available. `totalTokens` is the exact `prompt_tokens + completion_tokens` aggregate and is omitted if a supplied `total_tokens` disagrees.
 
 #### KV Cache effect
 

+ 2 - 2
packages/llm/llm-deepseek/README.zh.md

@@ -58,7 +58,7 @@ kind: "package-reference"
 | `defaultContextWindow` | `1,000,000` | 无精确值模型的容量回退 |
 | `models` | V4 Flash + V4 Pro + V4 Flash Vision Exp | 供发现消费方查看的建议性目录 |
 | `streamIdleTimeoutMs` | `300,000` | 单次流读取未完成的最大提供方空闲时间 |
-| `maxRequestFilesBytes` | `128 MiB` | 按最旧优先卸载前保留的请求图片字节高水位 |
+| `maxRequestFilesBytes` | `128 MiB` | 向 agent loop 的 `image/offload` 水位声明的 file 模式请求图片字节预算 |
 | `maxInlineRequestImageBytes` | `20 MiB` | 独立的 base64 回退高水位 |
 | `maxImagesPerRequest` | `600` | 保留请求图片数量的高水位 |
 | `imageOffloadByteQuantum` | `64 MiB` | Files 模式最旧前缀移除量子 |
@@ -156,7 +156,7 @@ Files 模式通过 `maxRequestFilesBytes` 与 `maxImagesPerRequest` 限制保留
 
 #### Token 影响
 
-提供方分词决定精确的文本与图片 token 输入。适配器声明按路由的 `imageRequestPricing`:它根据持久字节长度复现最旧优先的图片 offload,并按投影后的尺寸使用官方公布的 v4 视觉计量(14px patch 网格、3:1 降采样、单图 384 token 上限、最坏对齐 pad)为每张保留图片计价。这使 token 计量服务可以在请求发出前为图片压力定价;上报的 usage 仍是权威值。推理回传会把每个推理轮次的思维链带进后续请求,而丢弃超预算图片会避免再次为它们付费。可用时报告缓存读取用量。`totalTokens` 是精确的 `prompt_tokens + completion_tokens` 汇总值;提供方给出的 `total_tokens` 不一致时省略该值。
+提供方分词决定精确的文本与图片 token 输入。适配器声明按路由的 `imageRequestPricing`:把会话 `image/offload` 水位标记为已省略的每个出现位置按其占位文本计价,并按投影后的尺寸使用官方公布的 v4 视觉计量(14px patch 网格、3:1 降采样、单图 384 token 上限、最坏对齐 pad)为每张保留图片计价。这使 token 计量服务可以在请求发出前为图片压力定价;上报的 usage 仍是权威值。推理回传会把每个推理轮次的思维链带进后续请求,而已省略的图片不再消耗视觉 token。每条支持图片的目录路由还会把 file 模式预算(`maxRequestFilesBytes`、`maxImagesPerRequest`、两个量子与请求版本字节目标)作为 `imageRequest` 声明在其解析后的模型信息上,供 agent loop 仅凭已记录事实推进水位;保留的出现位置按精确请求版本字节仍超过 file 模式或内联回退预算的请求,以 `IMAGE_OFFLOAD_REQUIRED` 失败并说明还需省略多少最老的出现位置。可用时报告缓存读取用量。`totalTokens` 是精确的 `prompt_tokens + completion_tokens` 汇总值;提供方给出的 `total_tokens` 不一致时省略该值。
 
 #### KV Cache 影响
 

+ 25 - 16
packages/llm/llm-deepseek/src/adapter.ts

@@ -8,7 +8,7 @@
  * @module dsh-llm-deepseek/adapter
  */
 
-import { attributionHeaders, contentHasImage, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, offloadedImageText, offloadRequestImagesWithPolicy, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId } from '@deepseek-ai/dsh-llm'
+import { attributionHeaders, contentHasImage, CONTEXT_WINDOW_EXCEEDED_CODE, isContextWindowExceededError, isQuotaExceededError, LlmAdapter, LlmError, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId } from '@deepseek-ai/dsh-llm'
 import type {
   ContentBlock,
   GenerateOptions,
@@ -19,8 +19,7 @@ import type {
   LlmResolvedModelInfo,
   ModelModality,
   ResolvedRetryPolicy,
-  StreamChunk,
-} from '@deepseek-ai/dsh-llm'
+  StreamChunk, LlmImageRequestBudget } from '@deepseek-ai/dsh-llm'
 import type {
   AttachmentId,
   AttachmentStore,
@@ -205,8 +204,23 @@ function collectImageRefs(
   refs: Map<AttachmentId, ImageAttachmentRef>,
 ): void {
   for (const block of content) {
-    if (block.type === 'image') refs.set(block.attachment.attachmentId, block.attachment)
-    else if (block.type === 'tool-result') collectImageRefs(block.content, refs)
+    if (block.type === 'image') {
+      if (block.offloaded !== true) refs.set(block.attachment.attachmentId, block.attachment)
+    } else if (block.type === 'tool-result') {
+      collectImageRefs(block.content, refs)
+    }
+  }
+}
+
+/** The file-mode request-image budget one image-capable catalog route declares to the agent loop. */
+function imageRequestBudget(connection: DeepSeekConnectionOptions, model: DeepSeekCatalogModel): LlmImageRequestBudget {
+  return {
+    representation: 'raw',
+    maxBytes: connection.maxRequestFilesBytes,
+    maxImages: connection.maxImagesPerRequest,
+    byteQuantum: connection.imageOffloadByteQuantum,
+    countQuantum: connection.imageOffloadCountQuantum,
+    versionMaxBytes: resolveRequestImagePolicy(model).maxBytes,
   }
 }
 
@@ -407,6 +421,9 @@ export class DeepSeekAdapter extends LlmAdapter {
         : modelInfo(provider, configured),
       context: { contextWindow },
       defaultMaxTokens: configured?.maxTokens ?? connection.maxTokens,
+      ...configured?.inputModalities?.includes('image') === true
+        ? { imageRequest: imageRequestBudget(connection, configured) }
+        : {},
       ...connection.defaults.thinking === 'disabled'
         ? {
           reasoning: {
@@ -544,21 +561,13 @@ export class DeepSeekAdapter extends LlmAdapter {
 
     const fileConnection = { baseURL: connection.baseURL, apiKey }
     const model = connection.models.find(entry => entry.id === options.model)
-    const policy = model === undefined ? undefined : resolveRequestImagePolicy(model)
     const resolveImageAccess = attachments === undefined
       ? undefined
       : (ref: ImageAttachmentRef): ImageAttachmentAccess | undefined => this.config.resolveImageAccess?.(attachments, ref)
     const imageAccessOptions = resolveImageAccess === undefined ? {} : { resolveImageAccess }
-    const requestMessages = policy === undefined ? options.messages : offloadRequestImagesWithPolicy(options.messages, {
-      representation: 'raw',
-      maxBytes: connection.maxRequestFilesBytes,
-      maxImages: connection.maxImagesPerRequest,
-      byteQuantum: connection.imageOffloadByteQuantum,
-      countQuantum: connection.imageOffloadCountQuantum,
-      byteLength: ref => Math.min(ref.bytes, policy.maxBytes),
-      placeholder: ref => offloadedImageText(ref, resolveImageAccess?.(ref)),
-    })
-    const requestOptions = requestMessages === options.messages ? options : { ...options, messages: [...requestMessages] }
+    // The offloaded set is a durable surface fact carried by each image block;
+    // serialization projects it and only retained occurrences are prepared.
+    const requestOptions = options
     const requestImages = attachments === undefined || model === undefined
       ? new Map<AttachmentId, RequestImageAttachment>()
       : await prepareRequestImages(requestOptions, attachments, model, signal)

+ 26 - 39
packages/llm/llm-deepseek/src/request-pricing.ts

@@ -1,18 +1,18 @@
 /**
- * Provider-side request-image pricing for DeepSeek routes: reproduces the
- * adapter's deterministic request projection (per-model pixel budget,
- * oldest-first offload under the raw-byte and count budgets) and prices every
- * retained image with the published v4 vision-token accounting. Consumed
- * synchronously by the token meter through `LlmAdapter.imageRequestPricing`;
- * provider usage remains the authoritative anchor for completed requests.
+ * Provider-side request-image pricing for DeepSeek routes: prices every
+ * retained surface occurrence at its per-model pixel-budget projection with
+ * the published v4 vision-token accounting, and every occurrence the surface
+ * marks offloaded as its placeholder text. Consumed synchronously by the
+ * token meter through `LlmAdapter.imageRequestPricing`; provider usage
+ * remains the authoritative anchor for completed requests.
  *
  * @module dsh-llm-deepseek/request-pricing
  */
 
-import { offloadedImageText, offloadedImagePrefixCount, requestImageHandleText, textOnlyImageText } from '@deepseek-ai/dsh-llm'
-import type { ImageAttachmentAccessResolver, LlmImageRequestPrice, LlmImageRequestPricing } from '@deepseek-ai/dsh-llm'
+import { offloadedImageText, requestImageHandleText, textOnlyImageText } from '@deepseek-ai/dsh-llm'
+import type { ImageAttachmentAccessResolver, ImageBlock, LlmImageRequestPrice, LlmImageRequestPricing } from '@deepseek-ai/dsh-llm'
 import { requestImageDimensions } from '@deepseek-ai/dsh-attachment'
-import type { ImageAttachmentRef, ImageRequestPolicy } from '@deepseek-ai/dsh-attachment'
+import type { ImageRequestPolicy } from '@deepseek-ai/dsh-attachment'
 import { deepSeekImageTokens } from './image-tokens.ts'
 import type { DeepSeekCatalogModel, DeepSeekConnectionOptions } from './adapter.ts'
 
@@ -50,21 +50,19 @@ export function resolveRequestImagePolicy(model: DeepSeekCatalogModel): ImageReq
  * reproducing the `projectImagesForTextModel` substitution `LlmRuntime`
  * applies before dispatching to a route without the `image` modality.
  */
-function textOnlyPrice(ref: ImageAttachmentRef): LlmImageRequestPrice {
-  return { visualTokens: 0, text: textOnlyImageText(ref) }
+function textOnlyPrice(block: ImageBlock): LlmImageRequestPrice {
+  return { visualTokens: 0, text: textOnlyImageText(block.attachment) }
 }
 
 /**
  * Build the request-image pricing for one DeepSeek route from a validated
  * connection snapshot. Uncatalogued and text-only models price every
  * occurrence as its deterministic text substitution; image-capable models
- * reproduce the adapter's first-stage oldest-first offload from durable byte
- * lengths and price retained images by their projected request dimensions,
- * with each occurrence's handle or placeholder text built through the same
- * access resolution the serializer uses. The base64 fallback's tighter inline
- * budget is not reproduced, so a fallback request can only cost less than
- * this estimate; access paths resolve at pricing time, so a path that changes
- * before the request only shifts the text price by its own length.
+ * price an offloaded occurrence as its placeholder text and a retained one by
+ * its projected request dimensions, with each occurrence's handle or
+ * placeholder text built through the same access resolution the serializer
+ * uses. Access paths resolve at pricing time, so a path that changes before
+ * the request only shifts the text price by its own length.
  * @param connection - validated connection facts of the pricing resolution.
  * @param model - exact model id named by the request header.
  * @param resolveAccess - current execution-world access resolution shared with request serialization.
@@ -81,26 +79,15 @@ export function deepSeekImageRequestPricing(
   }
   const policy = resolveRequestImagePolicy(catalogModel)
   return {
-    priceImages: (images) => {
-      const offloaded = offloadedImagePrefixCount(
-        images.map(ref => Math.min(ref.bytes, policy.maxBytes)),
-        {
-          maxBytes: connection.maxRequestFilesBytes,
-          maxImages: connection.maxImagesPerRequest,
-          byteQuantum: connection.imageOffloadByteQuantum,
-          countQuantum: connection.imageOffloadCountQuantum,
-        },
-      )
-      return images.map((ref, index) => {
-        if (index < offloaded) {
-          return { visualTokens: 0, text: offloadedImageText(ref, resolveAccess?.(ref)) }
-        }
-        const dimensions = requestImageDimensions(ref.width, ref.height, policy.maxPixels)
-        return {
-          visualTokens: deepSeekImageTokens(dimensions.width, dimensions.height),
-          text: requestImageHandleText(ref, dimensions, resolveAccess?.(ref)),
-        }
-      })
-    },
+    priceImages: images => images.map(({ attachment: ref, offloaded }) => {
+      if (offloaded === true) {
+        return { visualTokens: 0, text: offloadedImageText(ref, resolveAccess?.(ref)) }
+      }
+      const dimensions = requestImageDimensions(ref.width, ref.height, policy.maxPixels)
+      return {
+        visualTokens: deepSeekImageTokens(dimensions.width, dimensions.height),
+        text: requestImageHandleText(ref, dimensions, resolveAccess?.(ref)),
+      }
+    }),
   }
 }

+ 43 - 19
packages/llm/llm-deepseek/src/serialize.ts

@@ -6,7 +6,7 @@
  * @module dsh-llm-deepseek/serialize
  */
 
-import { contentHasImage, LlmError, offloadedImageText, offloadRequestImagesWithPolicy, requestImageHandleText } from '@deepseek-ai/dsh-llm'
+import { contentHasImage, IMAGE_OFFLOAD_REQUIRED_CODE, LlmError, offloadedImageText, offloadedImagePrefixCount, projectOffloadedImages, representedImageBytes, requestImageHandleText, visitImageBlocks } from '@deepseek-ai/dsh-llm'
 import type { ContentBlock, GenerateOptions, ImageAttachmentAccessResolver, Message } from '@deepseek-ai/dsh-llm'
 import type { ImageAttachmentRef, RequestImageAttachment } from '@deepseek-ai/dsh-attachment'
 import type {
@@ -278,7 +278,7 @@ export function serializeMessages(messages: Message[]): WireMessage[] {
  * Serialize image-capable history after resolving durable attachments.
  * Consecutive tool results keep string `tool` messages and share one following
  * user message containing their images.
- * @param messages - transient request history after request-size offloading.
+ * @param messages - request history whose offloaded occurrences are already placeholder text.
  * @param images - prepared request versions, one provider representation, and its budget.
  * @returns ordered DeepSeek wire messages.
  */
@@ -391,10 +391,44 @@ export function serializeRequest(
   return requestWithMessages(options, messages, defaults)
 }
 
+/**
+ * Reject a request whose retained occurrences, at their exact request-version
+ * byte lengths under this representation, still exceed the route budget. The
+ * failure names how many more of the oldest retained occurrences the agent
+ * loop must offload; the surface, not this serializer, owns the offloaded set.
+ */
+function assertRetainedImagesFit(messages: readonly Message[], images: ImageSerializationOptions): void {
+  const representation = images.representation.kind === 'file' ? 'raw' : 'base64'
+  const lengths: number[] = []
+  for (const message of messages) {
+    visitImageBlocks(message.content, (block) => {
+      if (block.offloaded === true) return
+      const version = images.requestImages.get(block.attachment.attachmentId)
+      if (version === undefined) {
+        throw new LlmError(`DeepSeek request image ${block.attachment.attachmentId} was not prepared.`, 'INVALID_REQUEST')
+      }
+      lengths.push(representedImageBytes(version.bytes, { representation }))
+    })
+  }
+  const offloadImages = offloadedImagePrefixCount(lengths, {
+    maxBytes: images.maxRequestImageBytes,
+    ...images.maxImagesPerRequest === undefined ? {} : { maxImages: images.maxImagesPerRequest },
+    ...images.byteQuantum === undefined ? {} : { byteQuantum: images.byteQuantum },
+    ...images.countQuantum === undefined ? {} : { countQuantum: images.countQuantum },
+  })
+  if (offloadImages > 0) {
+    throw new LlmError(
+      `DeepSeek ${representation} request images exceed the route budget; ${offloadImages} more oldest occurrence(s) must be offloaded.`,
+      IMAGE_OFFLOAD_REQUIRED_CODE,
+      { offloadImages },
+    )
+  }
+}
+
 /**
  * Build one image-capable request while keeping durable bytes out of session
- * messages. Oversized oldest images become per-image text after their
- * exact request-version byte lengths are known and before provider serialization.
+ * messages. Offloaded occurrences become per-image text; retained occurrences
+ * must fit the route budget at their exact request-version byte lengths.
  * @param options - harness request containing image-capable user content.
  * @param images - request versions, optional current access resolver, and request bounds.
  * @param defaults - adapter-level thinking defaults.
@@ -406,21 +440,11 @@ export async function serializeRequestWithImages(
   defaults: RequestDefaults = {},
 ): Promise<WireRequest> {
   assertSupportedImageRoles(options.messages)
-  const requestMessages = offloadRequestImagesWithPolicy(options.messages, {
-    representation: images.representation.kind === 'file' ? 'raw' : 'base64',
-    byteLength: (ref) => {
-      const version = images.requestImages.get(ref.attachmentId)
-      if (version === undefined) {
-        throw new LlmError(`DeepSeek request image ${ref.attachmentId} was not prepared.`, 'INVALID_REQUEST')
-      }
-      return version.bytes
-    },
-    maxBytes: images.maxRequestImageBytes,
-    ...images.maxImagesPerRequest === undefined ? {} : { maxImages: images.maxImagesPerRequest },
-    ...images.byteQuantum === undefined ? {} : { byteQuantum: images.byteQuantum },
-    ...images.countQuantum === undefined ? {} : { countQuantum: images.countQuantum },
-    placeholder: ref => offloadedImageText(ref, images.resolveImageAccess?.(ref)),
-  })
+  assertRetainedImagesFit(options.messages, images)
+  const requestMessages = projectOffloadedImages(
+    options.messages,
+    ref => offloadedImageText(ref, images.resolveImageAccess?.(ref)),
+  )
   const messages: WireMessage[] = []
   if (options.system !== undefined) {
     messages.push({ role: 'system', content: options.system })

+ 33 - 15
packages/llm/llm-deepseek/tests/adapter.spec.ts

@@ -163,10 +163,10 @@ describe('request image policy', () => {
     const adapter = adapterOf({
       models: [{ id: 'vision', inputModalities: ['text', 'image'] }],
     })
-    const priced = adapter.imageRequestPricing('deepseek-official', 'vision')?.priceImages([imageRef])
+    const priced = adapter.imageRequestPricing('deepseek-official', 'vision')?.priceImages([{ type: 'image', attachment: imageRef }])
     expect(priced).toHaveLength(1)
     expect(priced?.[0]!.visualTokens).toBeGreaterThan(0)
-    const textOnly = adapter.imageRequestPricing('deepseek-official', 'unlisted')?.priceImages([imageRef])
+    const textOnly = adapter.imageRequestPricing('deepseek-official', 'unlisted')?.priceImages([{ type: 'image', attachment: imageRef }])
     expect(textOnly?.[0]!.visualTokens).toBe(0)
   })
 
@@ -182,7 +182,7 @@ describe('request image policy', () => {
         : undefined),
       prepareExtensions: noExtensions,
     })
-    const priced = adapter.imageRequestPricing('deepseek-official', 'vision')?.priceImages([imageRef])
+    const priced = adapter.imageRequestPricing('deepseek-official', 'vision')?.priceImages([{ type: 'image', attachment: imageRef }])
     expect(priced?.[0]?.text).toContain('/world/img.png')
   })
 })
@@ -441,7 +441,7 @@ describe('DeepSeekAdapter against a mock server', () => {
     expect(files.ensureUploaded).toHaveBeenCalledTimes(1)
   })
 
-  it('reduces base64 fallback history from the configured high watermark to its half-size quantum', async () => {
+  it('fails a base64 fallback whose retained images exceed the inline bound with the count to offload', async () => {
     const server = await mockServer([{ kind: 'sse', events: textEvents }])
     const attachments = attachmentStoreOf(ref => Promise.resolve(requestImage(ref))).store
     const files = fileStoreOf(() => Promise.reject(new LlmError('Files unavailable', 'SERVER')))
@@ -452,20 +452,16 @@ describe('DeepSeekAdapter against a mock server', () => {
       models: [{ id: 'deepseek-v4-flash-vision-exp', inputModalities: ['text', 'image'] }],
     }, attachments, files.store)
 
-    await drain(adapter.stream({
+    await expect(drain(adapter.stream({
       provider: 'deepseek-official',
       model: 'deepseek-v4-flash-vision-exp',
       messages: [createUserMessage({
         content: Array.from({ length: 21 }, () => ({ type: 'image' as const, attachment: imageRef })),
         source: { kind: 'plugin', plugin: 'test' },
       })],
-    }))
-
-    const body = JSON.stringify(server.requests[0])
-    expect(body.match(/image omitted to fit request image limits/g)).toHaveLength(11)
-    expect(body.match(/"type":"image_url"/g)).toHaveLength(10)
+    }))).rejects.toMatchObject({ code: 'IMAGE_OFFLOAD_REQUIRED', message: expect.stringContaining('11 more') as string })
+    expect(server.requests).toHaveLength(0)
   })
-
   it('discards partially resolved file ids and falls back with every retained image inline', async () => {
     const server = await mockServer([{ kind: 'sse', events: textEvents }])
     const secondRef = { ...imageRef, attachmentId: AttachmentId(`sha256:${'c'.repeat(64)}`) }
@@ -592,7 +588,7 @@ describe('DeepSeekAdapter against a mock server', () => {
     expect(JSON.stringify(server.requests[0])).not.toContain('image_url')
   })
 
-  it('does not prepare an old image removed by request offload', async () => {
+  it('does not prepare a surface-offloaded image and sends its placeholder instead', async () => {
     const server = await mockServer([{ kind: 'sse', events: textEvents }])
     const old = { ...imageRef, attachmentId: AttachmentId(`sha256:${'c'.repeat(64)}`), bytes: 3 }
     const recent = { ...imageRef, attachmentId: AttachmentId(`sha256:${'d'.repeat(64)}`), bytes: 3 }
@@ -603,8 +599,6 @@ describe('DeepSeekAdapter against a mock server', () => {
     const adapter = adapterOf({
       baseURL: server.url,
       models: [{ id: 'deepseek-v4-flash-vision-exp', inputModalities: ['text', 'image'] }],
-      maxRequestFilesBytes: 4,
-      imageOffloadByteQuantum: 2,
     }, attachmentMocks.store)
 
     await drain(adapter.stream({
@@ -612,7 +606,7 @@ describe('DeepSeekAdapter against a mock server', () => {
       model: 'deepseek-v4-flash-vision-exp',
       messages: [createUserMessage({
         content: [
-          { type: 'image', attachment: old },
+          { type: 'image', attachment: old, offloaded: true },
           { type: 'image', attachment: recent },
         ],
         source: { kind: 'plugin', plugin: 'test' },
@@ -635,6 +629,30 @@ describe('DeepSeekAdapter against a mock server', () => {
     })
   })
 
+  it('declares the file-mode request-image budget only for image-capable catalog routes', async () => {
+    const adapter = adapterOf({
+      maxRequestFilesBytes: 4096,
+      maxImagesPerRequest: 40,
+      imageOffloadByteQuantum: 1024,
+      imageOffloadCountQuantum: 20,
+      models: [
+        { id: 'vision', inputModalities: ['text', 'image'], imageMaxBytes: 2048 },
+        { id: 'text-only' },
+      ],
+    })
+    await expect(adapter.resolveModel('deepseek-official', 'vision')).resolves.toMatchObject({
+      imageRequest: {
+        representation: 'raw',
+        maxBytes: 4096,
+        maxImages: 40,
+        byteQuantum: 1024,
+        countQuantum: 20,
+        versionMaxBytes: 2048,
+      },
+    })
+    const textOnly = await adapter.resolveModel('deepseek-official', 'text-only')
+    expect(textOnly.imageRequest).toBeUndefined()
+  })
   it('projects nested tool-result images with route-owned request budgets', async () => {
     const server = await mockServer([
       { kind: 'sse', events: textEvents },

+ 20 - 2
packages/llm/llm-deepseek/tests/dynamic-config.spec.ts

@@ -219,16 +219,34 @@ describe('request-level dynamic configuration', () => {
 
     await assemble(ctx, { model: 'deepseek-v4-flash-vision-exp', messages })
     await ctx.settings.update(NS, { maxRequestFilesBytes: 4, imageOffloadByteQuantum: 2 })
-    await assemble(ctx, { model: 'deepseek-v4-flash-vision-exp', messages })
+    // The tightened budget reaches the agent loop through the resolved route metadata...
+    await expect(ctx.llm.resolveModelInfo('deepseek-official', 'deepseek-v4-flash-vision-exp'))
+      .resolves.toMatchObject({ imageRequest: { representation: 'raw', maxBytes: 4, byteQuantum: 2 } })
+    // ...and a request whose retained exact bytes still exceed it names the occurrences to offload.
+    const rejected = await assemble(ctx, { model: 'deepseek-v4-flash-vision-exp', messages })
+    expect(rejected.finish).toMatchObject({
+      kind: 'error',
+      failure: { code: 'IMAGE_OFFLOAD_REQUIRED', offloadImages: 1 },
+    })
+    await assemble(ctx, {
+      model: 'deepseek-v4-flash-vision-exp',
+      messages: [createUserMessage({
+        content: [
+          { type: 'image', attachment: IMAGE_REF, offloaded: true },
+          { type: 'image', attachment: IMAGE_REF },
+        ],
+        source: { kind: 'plugin', plugin: 'test' },
+      })],
+    })
 
     const first = (server.requests[0] as { messages: Array<{ content: unknown }> }).messages[0]?.content
     const second = (server.requests[1] as { messages: Array<{ content: unknown }> }).messages[0]?.content
+    expect(server.requests).toHaveLength(2)
     expect(JSON.stringify(first).match(/"type":"file"/g)).toHaveLength(2)
     expect(JSON.stringify(second)).toContain('[image omitted to fit request image limits')
     expect(JSON.stringify(second)).toContain(MODEL_IMAGE_PATH)
     expect(JSON.stringify(second).match(/"type":"file"/g)).toHaveLength(1)
   })
-
   it('re-registers the route in place when the captured retry policy changes, without an empty-registry window', async () => {
     const dir = await home()
     const { ctx } = await boot(dir, { baseURL: 'http://127.0.0.1:1' })

+ 18 - 21
packages/llm/llm-deepseek/tests/request-pricing.spec.ts

@@ -1,5 +1,6 @@
 import { describe, expect, it } from 'vitest'
 import { offloadedImageText, requestImageHandleText, textOnlyImageText } from '@deepseek-ai/dsh-llm'
+import type { ImageBlock } from '@deepseek-ai/dsh-llm'
 import { AttachmentId } from '@deepseek-ai/dsh-attachment'
 import type { ImageAttachmentRef } from '@deepseek-ai/dsh-attachment'
 import { deepSeekImageRequestPricing } from '../src/request-pricing.ts'
@@ -22,6 +23,10 @@ function ref(name: string, width: number, height: number, bytes = 1024): ImageAt
   }
 }
 
+function block(attachment: ImageAttachmentRef, offloaded?: true): ImageBlock {
+  return { type: 'image', attachment, ...offloaded === undefined ? {} : { offloaded } }
+}
+
 function connection(config: Omit<Config, 'models'> = {}): ReturnType<typeof resolveAdapterOptions> {
   return resolveAdapterOptions(Object.assign({ models: [VISION_MODEL] }, config))
 }
@@ -29,20 +34,20 @@ function connection(config: Omit<Config, 'models'> = {}): ReturnType<typeof reso
 describe('DeepSeek request-image pricing', () => {
   it('prices an uncatalogued model as its text-only substitution', () => {
     const image = ref('photo', 1920, 1080)
-    const prices = deepSeekImageRequestPricing(connection(), 'unlisted').priceImages([image])
+    const prices = deepSeekImageRequestPricing(connection(), 'unlisted').priceImages([block(image)])
     expect(prices).toEqual([{ visualTokens: 0, text: textOnlyImageText(image) }])
   })
 
   it('prices a catalogued text-only model as its text-only substitution', () => {
     const image = ref('photo', 1920, 1080)
     const options = resolveAdapterOptions({ models: [{ id: 'text-only' }] })
-    const prices = deepSeekImageRequestPricing(options, 'text-only').priceImages([image])
+    const prices = deepSeekImageRequestPricing(options, 'text-only').priceImages([block(image)])
     expect(prices).toEqual([{ visualTokens: 0, text: textOnlyImageText(image) }])
   })
 
   it('prices a retained image by its projected request dimensions plus its handle text', () => {
     const image = ref('photo', 1920, 1080)
-    const prices = deepSeekImageRequestPricing(connection(), 'vision').priceImages([image])
+    const prices = deepSeekImageRequestPricing(connection(), 'vision').priceImages([block(image)])
     expect(prices).toEqual([{
       visualTokens: 369,
       text: requestImageHandleText(image, { width: 1066, height: 600 }),
@@ -54,7 +59,7 @@ describe('DeepSeek request-image pricing', () => {
     const options = resolveAdapterOptions({
       models: [{ ...VISION_MODEL, imagePixelBudget: 'low' as const }],
     })
-    const prices = deepSeekImageRequestPricing(options, 'vision').priceImages([image])
+    const prices = deepSeekImageRequestPricing(options, 'vision').priceImages([block(image)])
     expect(prices[0]!.visualTokens).toBe(201)
   })
 
@@ -62,10 +67,10 @@ describe('DeepSeek request-image pricing', () => {
     const access = { readonlyPath: '/world/attachments/photo.png' }
     const images = [ref('first', 800, 800), ref('second', 800, 800)]
     const prices = deepSeekImageRequestPricing(
-      connection({ maxImagesPerRequest: 1, imageOffloadCountQuantum: 1 }),
+      connection(),
       'vision',
       () => access,
-    ).priceImages(images)
+    ).priceImages([block(images[0]!, true), block(images[1]!)])
     expect(prices[0]).toEqual({ visualTokens: 0, text: offloadedImageText(images[0]!, access) })
     expect(prices[1]).toEqual({
       visualTokens: 349,
@@ -74,12 +79,12 @@ describe('DeepSeek request-image pricing', () => {
     expect(prices[1]?.text).toContain('/world/attachments/photo.png')
   })
 
-  it('prices count-offloaded oldest occurrences as their placeholder text', () => {
+  it('prices surface-offloaded occurrences as placeholder text and every retained one at its visual price', () => {
     const images = [ref('first', 800, 800), ref('second', 800, 800), ref('third', 800, 800)]
     const prices = deepSeekImageRequestPricing(
-      connection({ maxImagesPerRequest: 2, imageOffloadCountQuantum: 1 }),
+      connection({ maxImagesPerRequest: 1, imageOffloadCountQuantum: 1 }),
       'vision',
-    ).priceImages(images)
+    ).priceImages([block(images[0]!, true), block(images[1]!), block(images[2]!)])
     expect(prices).toEqual([
       { visualTokens: 0, text: offloadedImageText(images[0]!) },
       { visualTokens: 349, text: requestImageHandleText(images[1]!, { width: 800, height: 800 }) },
@@ -87,20 +92,12 @@ describe('DeepSeek request-image pricing', () => {
     ])
   })
 
-  it('caps each occurrence at the per-image byte target before the byte budget', () => {
-    // Each 5 MiB source counts as the 1 MiB request target, so a 2 MiB budget
-    // with a one-byte quantum removes exactly the oldest occurrence.
-    const oversized = 5 * 1024 * 1024
-    const images = [
-      ref('first', 800, 800, oversized),
-      ref('second', 800, 800, oversized),
-      ref('third', 800, 800, oversized),
-    ]
+  it('prices a retained oversized occurrence at its visual price: the surface, not the budget, decides offload', () => {
+    const oversized = ref('first', 800, 800, 5 * 1024 * 1024)
     const prices = deepSeekImageRequestPricing(
       connection({ maxRequestFilesBytes: 2 * 1024 * 1024, imageOffloadByteQuantum: 1 }),
       'vision',
-    ).priceImages(images)
-    expect(prices.map(price => price.visualTokens)).toEqual([0, 349, 349])
-    expect(prices[0]!.text).toBe(offloadedImageText(images[0]!))
+    ).priceImages([block(oversized)])
+    expect(prices.map(price => price.visualTokens)).toEqual([349])
   })
 })

+ 22 - 6
packages/llm/llm-deepseek/tests/serialize.spec.ts

@@ -589,7 +589,7 @@ describe('image serialization', () => {
     ])
   })
 
-  it('offloads oldest images before reads and keeps the newest image', async () => {
+  it('projects surface-offloaded occurrences to placeholders without resolving them', async () => {
     const resolveFileId = fileResolver()
     const png = imageRef('image/png', 3)
     const jpeg = imageRef('image/jpeg', 3)
@@ -601,7 +601,7 @@ describe('image serialization', () => {
       model: 'deepseek-v4-flash-vision-exp',
       messages: [createUserMessage({
         content: [
-          { type: 'image', attachment: png },
+          { type: 'image', attachment: png, offloaded: true },
           { type: 'image', attachment: jpeg },
         ],
         source: { kind: 'plugin', plugin: 'test' },
@@ -622,22 +622,38 @@ describe('image serialization', () => {
     expect(resolveFileId).toHaveBeenCalledTimes(1)
     expect(resolveFileId.mock.calls[0]?.[0]).toMatchObject({ attachment: { mediaType: 'image/jpeg' } })
   })
+  it('rejects retained inline images beyond the bound with the quantized count to offload', async () => {
+    const ref = imageRef('image/png', 3)
+    // 21 retained 3-byte images encode to 84 base64 bytes against an 80-byte bound with a
+    // 40-byte quantum: the removal crosses the quantum at the eleventh oldest occurrence.
+    await expect(serializeRequestWithImages(request({
+      model: 'deepseek-v4-flash-vision-exp',
+      messages: [createUserMessage({
+        content: Array.from({ length: 21 }, () => ({ type: 'image' as const, attachment: ref })),
+        source: { kind: 'plugin', plugin: 'test' },
+      })],
+    }), inlineImageOptions([ref], 80, 40))).rejects.toMatchObject({
+      code: 'IMAGE_OFFLOAD_REQUIRED',
+      failure: { offloadImages: 11 },
+    })
+  })
 
-  it('drops base64 history from a 20-unit high watermark to a 10-unit low watermark', async () => {
+  it('counts only retained occurrences against the bound', async () => {
     const ref = imageRef('image/png', 3)
     const wire = await serializeRequestWithImages(request({
       model: 'deepseek-v4-flash-vision-exp',
       messages: [createUserMessage({
-        content: Array.from({ length: 21 }, () => ({ type: 'image' as const, attachment: ref })),
+        content: [
+          ...Array.from({ length: 11 }, () => ({ type: 'image' as const, attachment: ref, offloaded: true as const })),
+          ...Array.from({ length: 10 }, () => ({ type: 'image' as const, attachment: ref })),
+        ],
         source: { kind: 'plugin', plugin: 'test' },
       })],
     }), inlineImageOptions([ref], 80, 40))
-
     const content = wire.messages[0]?.content
     expect(JSON.stringify(content).match(/image omitted to fit request image limits/g)).toHaveLength(11)
     expect(JSON.stringify(content).match(/"type":"image_url"/g)).toHaveLength(10)
   })
-
   it('rejects an unprepared image while computing exact request bytes', async () => {
     const ref = imageRef()
     await expect(serializeRequestWithImages(request({

+ 2 - 2
packages/llm/llm-pi-ai/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/llm/llm-pi-ai/README.md
-README.md: c0aacbe3e0b727cd7b6b163814dc96859b954b9c
-README.zh.md: 7b8cef7db2ba5cebe3e42345b308716ae0099c03
+README.md: da413435eb997f87c70f54ed74f3098c0baa437e
+README.zh.md: 5fb07af85243e33105980ab1f47a34cd6407b195

+ 3 - 3
packages/llm/llm-pi-ai/README.md

@@ -83,7 +83,7 @@ Each profile may set a `retryPolicy`; omission uses normal mode with five retrie
 | `defaultMaxTokens` | `32,768` | Output-cap fallback for undescribed models |
 | `requestImagePixelBudget` | `4,194,304` | Total-pixel budget for each deterministic request image |
 | `requestImageMaxBytes` | `1 MiB` | Encoded-byte target for each request image before base64 expansion |
-| `maxRequestImageBytes` | `20 MiB` | Aggregate base64 image-payload bound with oldest-first offload |
+| `maxRequestImageBytes` | `20 MiB` | Aggregate base64 image-payload bound declared to the agent loop's `image/offload` watermark |
 | `retryPolicy` | normal, 5 retries | Provider-owned retry policy executed by `dsh-llm-retry` |
 
 The generated [configuration catalog](../../../docs/config-catalog.md#deepseek-aidsh-llm-pi-ai) is the exhaustive source for every accepted field and its JSDoc.
@@ -174,7 +174,7 @@ Read these pages when the package-level contract is not enough. They move from t
 
 #### What the model sees
 
-The selected catalog model receives `GenerateOptions.system`, history, tools, and sampling fields supported by pi-ai's common streaming API. Each retained image is preceded by text naming its complete attachment id and actual request dimensions. When the current execution filesystem maps the attachment provider's host object, the text also carries a read-only normalized-object path and warns that normalization or request projection may have resized or re-encoded the upload. When accumulated base64 image payload exceeds the route's `maxRequestImageBytes`, each offloaded image keeps its own identity and currently resolved access in replacement text. Offloaded normalized attachments are not read or transformed. Provider-native replay metadata is restored only when the adapter validates it for the historical content.
+The selected catalog model receives `GenerateOptions.system`, history, tools, and sampling fields supported by pi-ai's common streaming API. Each retained image is preceded by text naming its complete attachment id and actual request dimensions. When the current execution filesystem maps the attachment provider's host object, the text also carries a read-only normalized-object path and warns that normalization or request projection may have resized or re-encoded the upload. Each occurrence the session's `image/offload` watermark marks offloaded keeps its own identity and currently resolved access in replacement text, and its normalized attachment is not read or transformed. When the retained occurrences' exact base64 payload still exceeds the route's `maxRequestImageBytes`, the call fails with `IMAGE_OFFLOAD_REQUIRED` so the agent loop advances the watermark and rebuilds the request. Provider-native replay metadata is restored only when the adapter validates it for the historical content.
 
 #### Token effect
 
@@ -182,7 +182,7 @@ Provider tokenization governs exact input. Retained images add the stable attach
 
 #### KV Cache effect
 
-Conversion preserves logical request order, while image handles and offload placeholders add model-visible text. A changed execution-world path rewrites a historical handle and can prevent reuse from that image even when attachment identity and request bytes stay stable. Changing adapter instance, provider, model, or another upstream token has the same suffix effect. Crossing the image bound replaces an earlier image with placeholder text, so reuse ends at that message until the offloaded prefix stabilizes.
+Conversion preserves logical request order, while image handles and offload placeholders add model-visible text. A changed execution-world path rewrites a historical handle and can prevent reuse from that image even when attachment identity and request bytes stay stable. Changing adapter instance, provider, model, or another upstream token has the same suffix effect. An `image/offload` advance replaces an earlier image with placeholder text, so reuse ends at that message; the watermark never retreats, so the prefix stays stable afterwards.
 
 ### Provider response
 

+ 3 - 3
packages/llm/llm-pi-ai/README.zh.md

@@ -83,7 +83,7 @@ kind: "package-reference"
 | `defaultMaxTokens` | `32,768` | 未描述模型的输出上限回退 |
 | `requestImagePixelBudget` | `4,194,304` | 每张确定性请求图片的总像素预算 |
 | `requestImageMaxBytes` | `1 MiB` | 每张请求图片在 base64 扩展前的编码字节目标 |
-| `maxRequestImageBytes` | `20 MiB` | 带最旧优先卸载的 base64 图片载荷总上限 |
+| `maxRequestImageBytes` | `20 MiB` | 向 agent loop 的 `image/offload` 水位声明的 base64 图片载荷总上限 |
 | `retryPolicy` | normal,5 次重试 | 由 `dsh-llm-retry` 执行的提供方自有重试策略 |
 
 生成的[配置目录](../../../docs/config-catalog.zh.md#deepseek-aidsh-llm-pi-ai)是每个受支持字段及其 JSDoc 的穷尽式真源。
@@ -174,7 +174,7 @@ pi-ai 不提供的路由需要 `api`、`baseURL` 与非空 `models` 列表;无
 
 #### 模型看到什么
 
-所选目录模型会收到 `GenerateOptions.system`、历史、工具与 pi-ai 通用流式 API 支持的采样字段。每张保留图片前都会有文本,注明其完整附件 id 与实际请求尺寸。当前执行文件系统可以映射附件提供方的宿主对象时,该文本还会携带只读规范化对象路径,并警告规范化或请求投影可能缩放或重新编码上传内容。当累计 base64 图片载荷超过路由的 `maxRequestImageBytes` 时,每张卸载图片都会在替换文本中保留自己的身份与当前已解析访问方式。卸载的规范化附件不会读取或变换。提供方原生回放元数据只在适配器针对历史内容校验通过后恢复。
+所选目录模型会收到 `GenerateOptions.system`、历史、工具与 pi-ai 通用流式 API 支持的采样字段。每张保留图片前都会有文本,注明其完整附件 id 与实际请求尺寸。当前执行文件系统可以映射附件提供方的宿主对象时,该文本还会携带只读规范化对象路径,并警告规范化或请求投影可能缩放或重新编码上传内容。会话 `image/offload` 水位标记为已卸载的每个出现位置都会在替换文本中保留自己的身份与当前已解析访问方式,其规范化附件不会读取或变换。当保留的出现位置按精确 base64 载荷仍超过路由的 `maxRequestImageBytes` 时,调用以 `IMAGE_OFFLOAD_REQUIRED` 失败,由 agent loop 推进水位并重建请求。提供方原生回放元数据只在适配器针对历史内容校验通过后恢复。
 
 #### Token 影响
 
@@ -182,7 +182,7 @@ pi-ai 不提供的路由需要 `api`、`baseURL` 与非空 `models` 列表;无
 
 #### KV Cache 影响
 
-转换保持逻辑请求顺序,图片句柄与卸载占位符则会添加模型可见文本。即使附件身份与请求字节保持稳定,执行世界路径变化也会改写历史句柄,并可能从该图片起阻止复用。更换适配器实例、提供方、模型或其他上游 token 具有相同的后缀影响。越过图片上限会把较早图片替换为占位文本,因此复用在该消息处结束,直到被卸载前缀稳定。
+转换保持逻辑请求顺序,图片句柄与卸载占位符则会添加模型可见文本。即使附件身份与请求字节保持稳定,执行世界路径变化也会改写历史句柄,并可能从该图片起阻止复用。更换适配器实例、提供方、模型或其他上游 token 具有相同的后缀影响。一次 `image/offload` 推进会把较早图片替换为占位文本,因此复用在该消息处结束;水位永不回退,此后前缀保持稳定。
 
 ### 提供方响应
 

+ 9 - 0
packages/llm/llm-pi-ai/src/adapter.ts

@@ -306,6 +306,15 @@ export class PiAiAdapter extends LlmAdapter {
       inputModalities: [...resolvedModel.input],
       context: { contextWindow: resolvedModel.contextWindow },
       ...configuredMaxTokens === undefined ? {} : { defaultMaxTokens: configuredMaxTokens },
+      ...resolvedModel.input.includes('image')
+        ? {
+          imageRequest: {
+            representation: 'base64' as const,
+            maxBytes: profile.maxRequestImageBytes,
+            versionMaxBytes: profile.requestImageMaxBytes,
+          },
+        }
+        : {},
       ...reasoningInfo(resolvedModel, defaultLevel),
     }
   }

+ 35 - 23
packages/llm/llm-pi-ai/src/context.ts

@@ -5,7 +5,7 @@
  */
 
 import { brandString } from '@deepseek-ai/dsh-brand'
-import { contentHasImage, LlmError, offloadedImageText, offloadRequestImagesWithPolicy, requestImageHandleText } from '@deepseek-ai/dsh-llm'
+import { contentHasImage, IMAGE_OFFLOAD_REQUIRED_CODE, LlmError, offloadedImageText, offloadedImagePrefixCount, projectOffloadedImages, representedImageBytes, requestImageHandleText, visitImageBlocks } from '@deepseek-ai/dsh-llm'
 import type { ContentBlock, GenerateOptions, ImageAttachmentAccessResolver, Message, ToolCallId } from '@deepseek-ai/dsh-llm'
 import type {
   AttachmentId,
@@ -94,8 +94,11 @@ function collectImageRefs(
   refs: Map<AttachmentId, ImageAttachmentRef>,
 ): void {
   for (const block of blocks) {
-    if (block.type === 'image') refs.set(block.attachment.attachmentId, block.attachment)
-    else if (block.type === 'tool-result') collectImageRefs(block.content, refs)
+    if (block.type === 'image') {
+      if (block.offloaded !== true) refs.set(block.attachment.attachmentId, block.attachment)
+    } else if (block.type === 'tool-result') {
+      collectImageRefs(block.content, refs)
+    }
   }
 }
 
@@ -192,7 +195,7 @@ export interface PiImageRequestContext {
   attachments: AttachmentStore
   /** Resolve current tool access separately from deterministic request-image versions. */
   resolveImageAccess: ImageAttachmentAccessResolver
-  /** Request-level bound on base64-encoded image payload; omission leaves every image in place. */
+  /** Request-level bound on the base64-encoded payload of retained images; omission leaves the bound unchecked. */
   maxRequestImageBytes?: number
   /** Route pixel and raw encoded-byte budgets. */
   requestImagePolicy?: ImageRequestPolicy
@@ -213,10 +216,11 @@ export function toPiContext(
 ): PiContext
 /**
  * Convert harness history to a pi-ai Context while resolving durable images.
- * Tool result names are recovered from preceding assistant tool calls. When
- * the accumulated base64 image payload exceeds `maxRequestImageBytes`, the
- * oldest images are replaced by text placeholders until the request fits, so
- * an image-heavy session keeps clearing gateway request-size caps.
+ * Tool result names are recovered from preceding assistant tool calls. Image
+ * occurrences the surface marks offloaded become text placeholders; when the
+ * retained occurrences' exact base64 payload still exceeds
+ * `maxRequestImageBytes`, the call fails with `IMAGE_OFFLOAD_REQUIRED` naming
+ * how many more oldest occurrences the agent loop must offload.
  * @param options - the harness request; `options.system` maps to pi-ai's single `systemPrompt` slot.
  * @param images - attachment provider, current path resolver, and request limits.
  * @param onReplayDegrade - forwarded to {@link toPiAssistant} for each assistant message.
@@ -248,21 +252,29 @@ async function toPiContextWithImages(
     maxBytes: DEFAULT_REQUEST_IMAGE_MAX_BYTES,
   }
   assertSupportedImageRoles(options.messages)
-  const requestMessages = offloadRequestImagesWithPolicy(options.messages, {
-    representation: 'base64',
-    ...maxRequestImageBytes === undefined ? {} : { maxBytes: maxRequestImageBytes },
-    byteQuantum: 1,
-    byteLength: ref => Math.min(ref.bytes, requestImagePolicy.maxBytes),
-    placeholder: ref => offloadedImageText(ref, resolveImageAccess(ref)),
-  })
-  const requestImages = await prepareRequestImages(requestMessages, attachments, requestImagePolicy, options.signal)
-  const exactMessages = offloadRequestImagesWithPolicy(requestMessages, {
-    representation: 'base64',
-    ...maxRequestImageBytes === undefined ? {} : { maxBytes: maxRequestImageBytes },
-    byteQuantum: 1,
-    byteLength: ref => (requestImages.get(ref.attachmentId) as RequestImageAttachment).bytes,
-    placeholder: ref => offloadedImageText(ref, resolveImageAccess(ref)),
-  })
+  const requestImages = await prepareRequestImages(options.messages, attachments, requestImagePolicy, options.signal)
+  if (maxRequestImageBytes !== undefined) {
+    const lengths: number[] = []
+    for (const message of options.messages) {
+      visitImageBlocks(message.content, (block) => {
+        if (block.offloaded === true) return
+        const version = requestImages.get(block.attachment.attachmentId) as RequestImageAttachment
+        lengths.push(representedImageBytes(version.bytes, { representation: 'base64' }))
+      })
+    }
+    const offloadImages = offloadedImagePrefixCount(lengths, { maxBytes: maxRequestImageBytes })
+    if (offloadImages > 0) {
+      throw new LlmError(
+        `pi-ai request images exceed the ${maxRequestImageBytes}-byte base64 bound; ${offloadImages} more oldest occurrence(s) must be offloaded.`,
+        IMAGE_OFFLOAD_REQUIRED_CODE,
+        { offloadImages },
+      )
+    }
+  }
+  const exactMessages = projectOffloadedImages(
+    options.messages,
+    ref => offloadedImageText(ref, resolveImageAccess(ref)),
+  )
   const toolNames = new Map<ToolCallId, string>()
   const messages: PiMessage[] = []
 

+ 28 - 26
packages/llm/llm-pi-ai/tests/context.spec.ts

@@ -257,20 +257,20 @@ describe('pi-ai request context conversion', () => {
     })
   })
 
-  it('replaces the oldest images with placeholders once the request payload bound is exceeded', async () => {
+  it('projects surface-offloaded occurrences to placeholders and prepares only retained ones', async () => {
     const readImageRequest = vi.fn((value: ImageAttachmentRef) => (
       Promise.resolve(requestImage(value, Uint8Array.of(1, 2, 3)))
     ))
     const store = projectionStore(readImageRequest)
     const sized: ImageAttachmentRef = { ...ref, bytes: 3 }
     const callId = ToolCallId('shot-call')
-    // Three 3-byte images cost 4 base64 characters each (12 total); a bound of
-    // 8 forces exactly the oldest one out, including one nested in a tool result.
+    // The nested occurrence is offloaded on the surface; the two retained 3-byte images cost
+    // 4 base64 characters each and fit the 8-byte bound exactly.
     const context = await toPiContext(request([
       user([{
         type: 'tool-result',
         toolCallId: callId,
-        content: [{ type: 'image', attachment: sized }],
+        content: [{ type: 'image', attachment: sized, offloaded: true }],
       }]),
       user([{ type: 'image', attachment: sized }, { type: 'text', text: 'newer' }]),
       user([{ type: 'image', attachment: sized }]),
@@ -305,8 +305,7 @@ describe('pi-ai request context conversion', () => {
     ])
     expect(readImageRequest).toHaveBeenCalledTimes(1)
   })
-
-  it('does not prepare an old image removed by the conservative request projection', async () => {
+  it('does not prepare a surface-offloaded image', async () => {
     const old = { ...ref, attachmentId: AttachmentId(`sha256:${'c'.repeat(64)}`), bytes: 3 }
     const recent = { ...ref, attachmentId: AttachmentId(`sha256:${'d'.repeat(64)}`), bytes: 3 }
     const readImageRequest = vi.fn((value: ImageAttachmentRef) => {
@@ -315,7 +314,7 @@ describe('pi-ai request context conversion', () => {
     })
 
     const context = await toPiContext(request([user([
-      { type: 'image', attachment: old },
+      { type: 'image', attachment: old, offloaded: true },
       { type: 'image', attachment: recent },
     ])]), imageContext(projectionStore(readImageRequest), { maxRequestImageBytes: 4 }))
 
@@ -330,16 +329,26 @@ describe('pi-ai request context conversion', () => {
     expect(readImageRequest).toHaveBeenCalledTimes(1)
     expect(readImageRequest.mock.calls[0]?.[0]).toEqual(recent)
   })
-
-  it('uses independently resolved access when exact encoded bytes require offload', async () => {
+  it('fails with the count to offload when exact encoded bytes exceed the bound', async () => {
     const sized: ImageAttachmentRef = { ...ref, bytes: 3 }
-    const access = { readonlyPath: '/tmp/dsh-normalized-image' }
     const readImageRequest = vi.fn((value: ImageAttachmentRef) => Promise.resolve({
       ...requestImage(value, Uint8Array.of(1, 2, 3, 4)),
     }))
 
-    const context = await toPiContext(request([
+    await expect(toPiContext(request([
       user([{ type: 'image', attachment: sized }]),
+    ]), imageContext(projectionStore(readImageRequest), { maxRequestImageBytes: 4 })))
+      .rejects.toMatchObject({ code: 'IMAGE_OFFLOAD_REQUIRED', failure: { offloadImages: 1 } })
+    expect(readImageRequest).toHaveBeenCalledTimes(1)
+  })
+
+  it('renders an offloaded occurrence with independently resolved access', async () => {
+    const sized: ImageAttachmentRef = { ...ref, bytes: 3 }
+    const access = { readonlyPath: '/tmp/dsh-normalized-image' }
+    const readImageRequest = vi.fn()
+
+    const context = await toPiContext(request([
+      user([{ type: 'image', attachment: sized, offloaded: true }]),
     ]), imageContext(projectionStore(readImageRequest), {
       maxRequestImageBytes: 4,
       resolveImageAccess: () => access,
@@ -350,10 +359,9 @@ describe('pi-ai request context conversion', () => {
       content: offloadedImageText(sized, access),
       timestamp: 0,
     }])
-    expect(readImageRequest).toHaveBeenCalledTimes(1)
+    expect(readImageRequest).not.toHaveBeenCalled()
   })
-
-  it('keeps every image at exactly the payload bound and drops all of them when even the newest cannot fit', async () => {
+  it('keeps every image at exactly the payload bound and rejects a single image that cannot fit', async () => {
     const sized: ImageAttachmentRef = { ...ref, bytes: 3 }
     const exact = await toPiContext(request([
       user([{ type: 'image', attachment: sized }]),
@@ -376,17 +384,12 @@ describe('pi-ai request context conversion', () => {
       Promise.resolve(requestImage(value, new Uint8Array(300)))
     ))
     const store = projectionStore(readImageRequest)
-    const oversized = await toPiContext(request([
+    await expect(toPiContext(request([
       user([{ type: 'image', attachment: { ...ref, bytes: 300 } }]),
-    ]), imageContext(store, { maxRequestImageBytes: 8 }))
-    // All-text content collapses to the string form; the placeholder still reaches the model.
-    expect(oversized.messages).toEqual([
-      { role: 'user', content: offloadedImageText({ ...ref, bytes: 300 }), timestamp: 0 },
-    ])
-    expect(readImageRequest).not.toHaveBeenCalled()
+    ]), imageContext(store, { maxRequestImageBytes: 8 })))
+      .rejects.toMatchObject({ code: 'IMAGE_OFFLOAD_REQUIRED', failure: { offloadImages: 1 } })
   })
-
-  it('offloads repeated image-block occurrences by position rather than shared object identity', async () => {
+  it('projects repeated image-block occurrences by their own surface mark', async () => {
     const sized: ImageAttachmentRef = { ...ref, bytes: 3 }
     const shared: ContentBlock = { type: 'image', attachment: sized }
     const readImageRequest = vi.fn((value: ImageAttachmentRef) => (
@@ -394,11 +397,11 @@ describe('pi-ai request context conversion', () => {
     ))
     const store = projectionStore(readImageRequest)
     const aliased = await toPiContext(
-      request([user([shared, shared])]),
+      request([user([{ ...shared, offloaded: true }, shared])]),
       imageContext(store, { maxRequestImageBytes: 4 }),
     )
     const replayed = await toPiContext(request([user([
-      { type: 'image', attachment: { ...sized } },
+      { type: 'image', attachment: { ...sized }, offloaded: true },
       { type: 'image', attachment: { ...sized } },
     ])]), imageContext(store, { maxRequestImageBytes: 4 }))
 
@@ -415,7 +418,6 @@ describe('pi-ai request context conversion', () => {
     expect(replayed.messages).toEqual(expected)
     expect(readImageRequest).toHaveBeenCalledTimes(2)
   })
-
   it('keeps empty text-only users while separating result-only messages', () => {
     const callId = ToolCallId('unknown-call')
     expect(toPiContext(request([

+ 2 - 2
packages/llm/llm/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/llm/llm/README.md
-README.md: 6baa020a9503f32a71231d30e78ec7a3d8862f2b
-README.zh.md: f985736da44a76b063f16915a53cb19979fe1962
+README.md: 749bd226b80cbcb232f8b2a7264a67d2bfa8815e
+README.zh.md: 501790afd6665e21683b39650b6c35241740437f

+ 3 - 3
packages/llm/llm/README.md

@@ -93,13 +93,13 @@ The service is built on one separation: **the logical contract is provider-neutr
 | [`src/call-config.ts`](src/call-config.ts) | Call-config validation, adapter-default materialization, and request freezing |
 | [`src/retry-policy.ts`](src/retry-policy.ts) | Provider-owned retry policy resolution (normal and always modes) |
 | [`src/error.ts`](src/error.ts) | `HarnessError`/`LlmError` taxonomy and provider-neutral failure codes |
-| [`src/content.ts`](src/content.ts) | Shared image-content helpers, including request-image offloading |
+| [`src/content.ts`](src/content.ts) | Shared image-content helpers: the image walk, offload watermark planning, and offloaded-image projection |
 | [`src/api-key.ts`](src/api-key.ts) | Credential format check shared by every adapter |
 | [`src/adapter-failure.ts`](src/adapter-failure.ts) | Failure normalization into terminal finish chunks |
 
 ### Main flow
 
-A request is validated against its exact model's capability — context window, output default, reasoning efforts, and input modalities — and any adapter-configured defaults are materialized, then the whole request is deep-frozen. `prepareCall()` binds those facts, detached context, and retry policy to the exact adapter generation that performs terminal dispatch, so HMR or dynamic settings cannot combine one generation's image capability with another generation's endpoint. An image-capable adapter projects durable references into route-specific request versions; `resolveImageAttachmentAccess()` separately maps an attachment provider's optional host object into the current tool execution world without changing the request image or its `variantId`. A text-only route receives deterministic per-image placeholders, including nested tool-result images, without rewriting append-only session history. `offloadRequestImagesWithPolicy()` removes oldest images deterministically by raw or base64 size and count or byte quanta; the pure `offloadedImagePrefixCount()` exposes that decision so route-owned request pricing can reproduce it without building the projection. Adapters that charge visual tokens declare per-route `imageRequestPricing`, which `ctx.llm.imageRequestPricing(provider, model)` resolves synchronously for the token meter. Dispatch goes through the `llm/stream` waterfall, then chunks return as token-level deltas and every adapter outcome reaches the consumer as one terminal `finish` chunk.
+A request is validated against its exact model's capability — context window, output default, reasoning efforts, and input modalities — and any adapter-configured defaults are materialized, then the whole request is deep-frozen. `prepareCall()` binds those facts, detached context, and retry policy to the exact adapter generation that performs terminal dispatch, so HMR or dynamic settings cannot combine one generation's image capability with another generation's endpoint. An image-capable adapter projects durable references into route-specific request versions; `resolveImageAttachmentAccess()` separately maps an attachment provider's optional host object into the current tool execution world without changing the request image or its `variantId`. A text-only route receives deterministic per-image placeholders, including nested tool-result images, without rewriting append-only session history. An image-capable route declares a `LlmImageRequestBudget` on its resolved model info; the agent loop plans the session's durable `image/offload` watermark from it with the pure `planImageOffload()`, surface derivation marks occurrences at or before the watermark `offloaded: true`, and `projectOffloadedImages()` renders those marks as route-owned placeholder text. An adapter whose exact byte accounting still exceeds its budget fails with `IMAGE_OFFLOAD_REQUIRED` naming the additional occurrences (`offloadedImagePrefixCount()`), never with an unlogged projection. Adapters that charge visual tokens declare per-route `imageRequestPricing`, which `ctx.llm.imageRequestPricing(provider, model)` resolves synchronously for the token meter. Dispatch goes through the `llm/stream` waterfall, then chunks return as token-level deltas and every adapter outcome reaches the consumer as one terminal `finish` chunk.
 
 ### Invariants
 
@@ -136,7 +136,7 @@ None, as the LLM service adds no content; adapters choose when to add the shared
 
 #### KV Cache effect
 
-Reasoning-effort materialization preserves the assembled request prefix. Image identity and request-preview text are deterministic, while an optional execution-world path is resolved for each request; a changed path or image-offload boundary can prevent reuse from that image.
+Reasoning-effort materialization preserves the assembled request prefix. Image identity and request-preview text are deterministic, while an optional execution-world path is resolved for each request; a changed path or an `image/offload` advance can prevent reuse from that image.
 
 ## Known Limitations and Deferred Work
 

+ 3 - 3
packages/llm/llm/README.zh.md

@@ -93,13 +93,13 @@ for await (const chunk of ctx.llm.stream({
 | [`src/call-config.ts`](src/call-config.ts) | 调用配置校验、适配器默认值填入与请求冻结 |
 | [`src/retry-policy.ts`](src/retry-policy.ts) | 提供方自有重试策略解析(normal 与 always 模式) |
 | [`src/error.ts`](src/error.ts) | `HarnessError`/`LlmError` 分类体系与提供方无关失败 code |
-| [`src/content.ts`](src/content.ts) | 共享图片内容辅助函数,包括请求图片卸载 |
+| [`src/content.ts`](src/content.ts) | 共享图片内容辅助函数:图片遍历、卸载水位规划与已卸载图片投影 |
 | [`src/api-key.ts`](src/api-key.ts) | 每个适配器共享的凭据格式校验 |
 | [`src/adapter-failure.ts`](src/adapter-failure.ts) | 把失败归一化为终止 finish 分片 |
 
 ### 主流程
 
-请求会对照其精确模型的能力——上下文窗口、输出默认值、推理强度与输入模态——校验,填入任何适配器配置的默认值,然后整个请求被深度冻结。`prepareCall()` 把这些事实、分离的上下文与重试策略绑定到执行最终分发的精确适配器代次,因此 HMR 或动态设置无法把一个代次的图片能力与另一代次的端点混用。支持图片的适配器把持久引用投影为路由专用请求版本;`resolveImageAttachmentAccess()` 会单独把附件提供方的可选宿主对象映射进当前工具执行世界,而不改变请求图片或其 `variantId`。纯文本路由接收确定性的逐图片占位符,包括嵌套工具结果图片,而不会改写仅追加会话历史。`offloadRequestImagesWithPolicy()` 按原始字节或 base64 大小以及图片数或字节步长,确定性地从最旧图片开始移除;纯函数 `offloadedImagePrefixCount()` 公开同一决策,使路由所属的请求定价无需构建投影即可复现它。对视觉 token 收费的适配器声明按路由的 `imageRequestPricing`,`ctx.llm.imageRequestPricing(provider, model)` 为 token meter 同步解析它。分发经过 `llm/stream` waterfall,随后分片以 token 级增量返回,每个适配器结果都以唯一一个终止 `finish` 分片到达消费方。
+请求会对照其精确模型的能力——上下文窗口、输出默认值、推理强度与输入模态——校验,填入任何适配器配置的默认值,然后整个请求被深度冻结。`prepareCall()` 把这些事实、分离的上下文与重试策略绑定到执行最终分发的精确适配器代次,因此 HMR 或动态设置无法把一个代次的图片能力与另一代次的端点混用。支持图片的适配器把持久引用投影为路由专用请求版本;`resolveImageAttachmentAccess()` 会单独把附件提供方的可选宿主对象映射进当前工具执行世界,而不改变请求图片或其 `variantId`。纯文本路由接收确定性的逐图片占位符,包括嵌套工具结果图片,而不会改写仅追加会话历史。支持图片的路由在其解析后的模型信息上声明 `LlmImageRequestBudget`;agent loop 用纯函数 `planImageOffload()` 据此规划会话的持久 `image/offload` 水位,表层派生把位于水位及之前的出现位置标为 `offloaded: true`,`projectOffloadedImages()` 把这些标记渲染为路由所属的占位文本。adapter 按精确字节计量仍超预算时,以 `IMAGE_OFFLOAD_REQUIRED` 失败并说明还需省略的出现位置数(`offloadedImagePrefixCount()`),绝不发送未记录的投影。对视觉 token 收费的适配器声明按路由的 `imageRequestPricing`,`ctx.llm.imageRequestPricing(provider, model)` 为 token meter 同步解析它。分发经过 `llm/stream` waterfall,随后分片以 token 级增量返回,每个适配器结果都以唯一一个终止 `finish` 分片到达消费方。
 
 ### 不变式
 
@@ -136,7 +136,7 @@ for await (const chunk of ctx.llm.stream({
 
 #### KV Cache 影响
 
-推理强度的具体化会保留已组装请求前缀。图片身份与请求预览文本是确定性的,可选执行世界路径则按请求解析;路径变化或图片卸载边界变化可能从该图片起阻止复用。
+推理强度的具体化会保留已组装请求前缀。图片身份与请求预览文本是确定性的,可选执行世界路径则按请求解析;路径变化或一次 `image/offload` 推进可能从该图片起阻止复用。
 
 ## 已知限制与延期工作
 

+ 4 - 1
packages/llm/llm/src/adapter-failure.ts

@@ -69,18 +69,21 @@ function failureSnapshot(value: unknown): LlmFailure | undefined {
     const status = candidate.status
     const providerRetryAfterMs = candidate.providerRetryAfterMs
     const requestId = candidate.requestId
+    const offloadImages = candidate.offloadImages
     if (typeof message !== 'string' || message.length === 0
       || typeof code !== 'string' || code.length === 0
       || (status !== undefined && (!Number.isInteger(status) || status < 100 || status > 599))
       || (providerRetryAfterMs !== undefined
         && (!Number.isFinite(providerRetryAfterMs) || providerRetryAfterMs <= 0))
-      || (requestId !== undefined && (typeof requestId !== 'string' || requestId.length === 0))) return undefined
+      || (requestId !== undefined && (typeof requestId !== 'string' || requestId.length === 0))
+      || (offloadImages !== undefined && (!Number.isSafeInteger(offloadImages) || offloadImages <= 0))) return undefined
     return Object.freeze({
       message,
       code,
       ...status === undefined ? {} : { status },
       ...providerRetryAfterMs === undefined ? {} : { providerRetryAfterMs },
       ...requestId === undefined ? {} : { requestId },
+      ...offloadImages === undefined ? {} : { offloadImages },
     })
   } catch (_sdkFailureGetter) {
     return undefined

+ 173 - 89
packages/llm/llm/src/content.ts

@@ -1,6 +1,6 @@
 /** Content-block structure helpers. @module @deepseek-ai/dsh-llm/content */
 
-import type { ContentBlock } from './types.ts'
+import type { ContentBlock, ImageBlock, LlmImageRequestBudget } from './types.ts'
 import type { Message } from './message.ts'
 import type { AttachmentStore, ImageAttachmentRef, ImageMediaType, RequestImageAttachment } from '@deepseek-ai/dsh-attachment'
 import { assertNever } from '@deepseek-ai/dsh-util-values'
@@ -131,80 +131,117 @@ function base64Length(bytes: number): number {
   return Math.ceil(bytes / 3) * 4
 }
 
-/** Byte accounting and quantized removal policy for one request representation. */
-export interface RequestImageOffloadPolicy {
-  /** Image count accepted by the route; omission leaves count unbounded. */
-  maxImages?: number
-  /** Accumulated image bytes accepted by the route; omission leaves bytes unbounded. */
-  maxBytes?: number
-  /** Number of excess images removed as one deterministic step. */
-  countQuantum?: number
-  /** Number of excess bytes removed as one deterministic step. */
-  byteQuantum?: number
-  /** Whether byte accounting uses raw file bytes or inline base64 length. */
-  representation: 'raw' | 'base64'
-  /** Resolve the encoded request-version length; omission uses normalized attachment bytes. */
-  byteLength?: (ref: ImageAttachmentRef) => number
-  /** Build the model-visible replacement for each omitted attachment. */
-  placeholder: (ref: ImageAttachmentRef) => string
+/**
+ * Block index path of one image occurrence inside message content: the
+ * top-level block index alone, or that index followed by the position inside
+ * a tool-result block's content. Paths compare lexicographically in message
+ * order.
+ */
+export type ImageBlockPath = readonly number[]
+
+/**
+ * Compare two block paths in message order.
+ * @param a - first path.
+ * @param b - second path.
+ * @returns negative, zero, or positive as `a` sorts before, at, or after `b`.
+ */
+export function compareImageBlockPaths(a: ImageBlockPath, b: ImageBlockPath): number {
+  const length = Math.min(a.length, b.length)
+  for (let index = 0; index < length; index += 1) {
+    // oxlint-disable-next-line typescript/no-non-null-assertion -- index is below both lengths
+    const delta = a[index]! - b[index]!
+    if (delta !== 0) return delta
+  }
+  return a.length - b.length
 }
 
-/** Collect represented image lengths in request and nested-block order. */
-function collectImageLengths(
-  blocks: readonly ContentBlock[],
-  lengths: number[],
-  policy: RequestImageOffloadPolicy,
+/**
+ * Visit every image occurrence of typed content in message order, including
+ * nested tool-result content, with the block path that identifies it. This is
+ * the one image walk shared by surface derivation, watermark planning, and
+ * pricing, so no consumer can diverge on nesting depth or occurrence order.
+ * @param content - typed model content blocks.
+ * @param visit - called once per occurrence with the block and its path.
+ */
+export function visitImageBlocks(
+  content: readonly ContentBlock[],
+  visit: (block: ImageBlock, path: ImageBlockPath) => void,
 ): void {
-  for (const block of blocks) {
+  for (const [index, block] of content.entries()) {
     if (block.type === 'image') {
-      const bytes = policy.byteLength === undefined
-        ? block.attachment.bytes
-        : policy.byteLength(block.attachment)
-      lengths.push(policy.representation === 'base64' ? base64Length(bytes) : bytes)
+      visit(block, [index])
     } else if (block.type === 'tool-result') {
-      collectImageLengths(block.content, lengths, policy)
+      for (const [nested, inner] of block.content.entries()) {
+        if (inner.type === 'image') visit(inner, [index, nested])
+      }
     }
   }
 }
 
-/** Replace the first `remaining.count` image occurrences without mutating durable messages. */
-function replaceOldestImages(
-  blocks: readonly ContentBlock[],
-  remaining: { count: number },
-  placeholder: (ref: ImageAttachmentRef) => string,
-): ContentBlock[] {
+/**
+ * Represented byte length of one image occurrence under a route budget: the
+ * normalized byte count clamped to the route's request-version target, then
+ * base64-expanded for an inline representation.
+ * @param bytes - normalized attachment byte count.
+ * @param budget - route representation and request-version target.
+ * @returns the byte length the route's request accounting charges.
+ */
+export function representedImageBytes(
+  bytes: number,
+  budget: Pick<LlmImageRequestBudget, 'representation' | 'versionMaxBytes'>,
+): number {
+  const clamped = budget.versionMaxBytes === undefined ? bytes : Math.min(bytes, budget.versionMaxBytes)
+  return budget.representation === 'base64' ? base64Length(clamped) : clamped
+}
+
+/**
+ * Mark image occurrences as offloaded without mutating durable content.
+ * @param content - typed model content blocks.
+ * @param offloaded - whether the occurrence at `path` lies at or before the watermark.
+ * @returns the original array when nothing changes, otherwise a copy whose marked blocks carry `offloaded: true`.
+ */
+export function markOffloadedImages(
+  content: readonly ContentBlock[],
+  offloaded: (path: ImageBlockPath) => boolean,
+): readonly ContentBlock[] {
   let next: ContentBlock[] | undefined
-  for (const [index, block] of blocks.entries()) {
-    if (block.type === 'image' && remaining.count > 0) {
-      remaining.count -= 1
-      next ??= blocks.slice(0, index)
-      next.push({ type: 'text', text: placeholder(block.attachment) })
-      continue
-    }
-    if (block.type === 'tool-result') {
-      const content = replaceOldestImages(block.content, remaining, placeholder)
-      if (content !== block.content) {
-        next ??= blocks.slice(0, index)
-        next.push({ ...block, content })
+  for (const [index, block] of content.entries()) {
+    if (block.type === 'image') {
+      if (block.offloaded !== true && offloaded([index])) {
+        next ??= content.slice(0, index)
+        next.push({ ...block, offloaded: true })
+        continue
+      }
+    } else if (block.type === 'tool-result') {
+      const inner = markOffloadedImages(
+        block.content,
+        path => offloaded([index, ...path]),
+      )
+      if (inner !== block.content) {
+        next ??= content.slice(0, index)
+        next.push({ ...block, content: inner as ContentBlock[] })
         continue
       }
     }
     next?.push(block)
   }
-  return next ?? blocks as ContentBlock[]
+  return next ?? content
 }
 
-/** Replace every image occurrence, including nested tool results, for a text-only model. */
-function replaceImagesForTextModel(blocks: readonly ContentBlock[]): ContentBlock[] {
+/** Replace every offloaded occurrence, including nested tool results, with its placeholder. */
+function replaceOffloadedImages(
+  blocks: readonly ContentBlock[],
+  placeholder: (ref: ImageAttachmentRef) => string,
+): ContentBlock[] {
   let next: ContentBlock[] | undefined
   for (const [index, block] of blocks.entries()) {
-    if (block.type === 'image') {
+    if (block.type === 'image' && block.offloaded === true) {
       next ??= blocks.slice(0, index)
-      next.push({ type: 'text', text: textOnlyImageText(block.attachment) })
+      next.push({ type: 'text', text: placeholder(block.attachment) })
       continue
     }
     if (block.type === 'tool-result') {
-      const content = replaceImagesForTextModel(block.content)
+      const content = replaceOffloadedImages(block.content, placeholder)
       if (content !== block.content) {
         next ??= blocks.slice(0, index)
         next.push({ ...block, content })
@@ -217,37 +254,43 @@ function replaceImagesForTextModel(blocks: readonly ContentBlock[]): ContentBloc
 }
 
 /**
- * Project durable image history into deterministic text for an exact text-only model.
- * @param messages - complete request history.
- * @returns the original list without images, otherwise shallow message copies with stable placeholders.
+ * Project the surface's offloaded occurrences into deterministic text for one
+ * request. The offloaded set is a durable surface fact, so every route sends
+ * the same set; only the placeholder text is route-owned.
+ * @param messages - derived request history.
+ * @param placeholder - build the model-visible replacement for one offloaded attachment.
+ * @returns the original list when nothing is offloaded, otherwise shallow message copies with placeholders.
  */
-export function projectImagesForTextModel(messages: readonly Message[]): readonly Message[] {
-  if (!messages.some(message => contentHasImage(message.content))) return messages
+export function projectOffloadedImages(
+  messages: readonly Message[],
+  placeholder: (ref: ImageAttachmentRef) => string,
+): readonly Message[] {
   return messages.map((message) => {
-    const content = replaceImagesForTextModel(message.content)
+    const content = replaceOffloadedImages(message.content, placeholder)
     return content === message.content ? message : { ...message, content }
   })
 }
 
 /**
- * Number of oldest image occurrences one request projection removes, in whole
- * count and byte quanta, once a route budget is exceeded. The result depends
- * only on the represented lengths, so provider request pricing reproduces the
- * exact serialization decision without building the projected messages.
- * @param lengths - represented byte length of every occurrence, in request order.
- * @param policy - count/byte budgets and removal quanta; unbounded when absent.
- * @returns how many leading occurrences the projection replaces with placeholders.
+ * Number of oldest retained image occurrences one route budget removes, in
+ * whole count and byte quanta, once the budget is exceeded. The result depends
+ * only on the represented lengths, so the agent loop plans the durable
+ * watermark from logged facts and an adapter names the same count when its
+ * exact accounting still overflows.
+ * @param lengths - represented byte length of every retained occurrence, oldest first.
+ * @param budget - count/byte budgets and removal quanta; unbounded when absent.
+ * @returns how many leading occurrences to offload.
  */
 export function offloadedImagePrefixCount(
   lengths: readonly number[],
-  policy: Pick<RequestImageOffloadPolicy, 'maxImages' | 'maxBytes' | 'countQuantum' | 'byteQuantum'>,
+  budget: Pick<LlmImageRequestBudget, 'maxImages' | 'maxBytes' | 'countQuantum' | 'byteQuantum'>,
 ): number {
   const total = lengths.reduce((sum, bytes) => sum + bytes, 0)
-  const excessCount = policy.maxImages === undefined ? 0 : Math.max(0, lengths.length - policy.maxImages)
-  const excessBytes = policy.maxBytes === undefined ? 0 : Math.max(0, total - policy.maxBytes)
+  const excessCount = budget.maxImages === undefined ? 0 : Math.max(0, lengths.length - budget.maxImages)
+  const excessBytes = budget.maxBytes === undefined ? 0 : Math.max(0, total - budget.maxBytes)
   if (excessCount === 0 && excessBytes === 0) return 0
-  const countQuantum = policy.countQuantum ?? 1
-  const byteQuantum = policy.byteQuantum ?? 1
+  const countQuantum = budget.countQuantum ?? 1
+  const byteQuantum = budget.byteQuantum ?? 1
   const removeCount = excessCount === 0 ? 0 : Math.ceil(excessCount / countQuantum) * countQuantum
   const removeBytes = excessBytes === 0 ? 0 : Math.ceil(excessBytes / byteQuantum) * byteQuantum
   let count = 0
@@ -262,28 +305,69 @@ export function offloadedImagePrefixCount(
   return count
 }
 
+/** One retained image occurrence the watermark planner may offload. */
+export interface RetainedImageOccurrence<Position> {
+  /** Durable position of the occurrence, used as the watermark when it becomes the last offloaded one. */
+  position: Position
+  /** Normalized attachment byte count. */
+  bytes: number
+}
+
 /**
- * Return a deterministic transient projection whose oldest images are replaced
- * in whole count and byte quanta after a route budget is exceeded. The target
- * depends only on complete durable history: at 129 one-megabyte images under
- * a 128 MiB bound with a 64 MiB quantum, the oldest 65 images are removed so
- * 64 MiB remain; that removed prefix stays fixed until total history exceeds
- * 192 MiB.
- * @param messages - complete request history, oldest first.
- * @param policy - route representation, budgets, and removal quanta.
- * @returns original messages below both bounds, otherwise shallow copies with deterministic placeholders.
+ * Plan the next durable watermark for one route budget: with the retained
+ * occurrences oldest first, the budget removes a whole-quantum prefix and the
+ * last removed occurrence's position becomes the watermark. Under a 128 MiB
+ * bound with a 64 MiB quantum, 129 retained one-megabyte images offload the
+ * oldest 65 so 64 MiB remain, and the watermark then holds until the retained
+ * total again exceeds the bound.
+ * @param retained - retained occurrences in log order, oldest first.
+ * @param budget - route representation, budgets, and removal quanta.
+ * @returns the position of the last occurrence to offload, or undefined when the budget holds.
  */
-export function offloadRequestImagesWithPolicy(
-  messages: readonly Message[],
-  policy: RequestImageOffloadPolicy,
-): readonly Message[] {
-  const lengths: number[] = []
-  for (const message of messages) collectImageLengths(message.content, lengths, policy)
-  const count = offloadedImagePrefixCount(lengths, policy)
-  if (count === 0) return messages
-  const remaining = { count }
+export function planImageOffload<Position>(
+  retained: readonly RetainedImageOccurrence<Position>[],
+  budget: LlmImageRequestBudget,
+): Position | undefined {
+  const count = offloadedImagePrefixCount(
+    retained.map(occurrence => representedImageBytes(occurrence.bytes, budget)),
+    budget,
+  )
+  if (count === 0) return undefined
+  // oxlint-disable-next-line typescript/no-non-null-assertion -- the prefix count never exceeds the retained length
+  return retained[count - 1]!.position
+}
+
+/** Replace every image occurrence, including nested tool results, for a text-only model. */
+function replaceImagesForTextModel(blocks: readonly ContentBlock[]): ContentBlock[] {
+  let next: ContentBlock[] | undefined
+  for (const [index, block] of blocks.entries()) {
+    if (block.type === 'image') {
+      next ??= blocks.slice(0, index)
+      next.push({ type: 'text', text: textOnlyImageText(block.attachment) })
+      continue
+    }
+    if (block.type === 'tool-result') {
+      const content = replaceImagesForTextModel(block.content)
+      if (content !== block.content) {
+        next ??= blocks.slice(0, index)
+        next.push({ ...block, content })
+        continue
+      }
+    }
+    next?.push(block)
+  }
+  return next ?? blocks as ContentBlock[]
+}
+
+/**
+ * Project durable image history into deterministic text for an exact text-only model.
+ * @param messages - complete request history.
+ * @returns the original list without images, otherwise shallow message copies with stable placeholders.
+ */
+export function projectImagesForTextModel(messages: readonly Message[]): readonly Message[] {
+  if (!messages.some(message => contentHasImage(message.content))) return messages
   return messages.map((message) => {
-    const content = replaceOldestImages(message.content, remaining, policy.placeholder)
+    const content = replaceImagesForTextModel(message.content)
     return content === message.content ? message : { ...message, content }
   })
 }

+ 9 - 0
packages/llm/llm/src/error.ts

@@ -161,3 +161,12 @@ export function errorChain(value: unknown): string {
 export function isHarnessError(value: unknown): value is HarnessError {
   return value instanceof HarnessError
 }
+
+/**
+ * Canonical code for a request an image-capable route cannot send until the
+ * agent loop advances the session's durable `image/offload` watermark. The
+ * failure's `offloadImages` names how many more of the oldest retained
+ * occurrences must be offloaded; the loop appends the advance and rebuilds the
+ * request, so the model never receives an unlogged projection.
+ */
+export const IMAGE_OFFLOAD_REQUIRED_CODE = 'IMAGE_OFFLOAD_REQUIRED'

+ 42 - 0
packages/llm/llm/src/index.ts

@@ -15,6 +15,7 @@ import type {
   LlmDiscoveredModel,
   LlmFailure,
   LlmImageRequestPricing,
+  LlmImageRequestBudget,
   LlmModelContext,
   LlmModelDiscoveryRequest,
   LlmModelInfo,
@@ -77,6 +78,8 @@ export interface LlmErrorOptions extends ErrorOptions {
   providerRetryAfterMs?: number
   /** Non-empty opaque provider request id. */
   requestId?: ProviderRequestId
+  /** Positive count of additional oldest retained image occurrences to offload; only with `IMAGE_OFFLOAD_REQUIRED`. */
+  offloadImages?: number
 }
 
 /**
@@ -107,6 +110,10 @@ export class LlmError extends HarnessError {
       && (typeof options.requestId !== 'string' || options.requestId.length === 0)) {
       throw new Error('LlmError requestId must be a non-empty string')
     }
+    if (options?.offloadImages !== undefined
+      && (!Number.isSafeInteger(options.offloadImages) || options.offloadImages <= 0)) {
+      throw new Error('LlmError offloadImages must be a positive safe integer')
+    }
     super(message, code, options)
     this.name = 'LlmError'
     this.failure = Object.freeze({
@@ -115,6 +122,7 @@ export class LlmError extends HarnessError {
       ...options?.status === undefined ? {} : { status: options.status },
       ...options?.providerRetryAfterMs === undefined ? {} : { providerRetryAfterMs: options.providerRetryAfterMs },
       ...options?.requestId === undefined ? {} : { requestId: options.requestId },
+      ...options?.offloadImages === undefined ? {} : { offloadImages: options.offloadImages },
     })
   }
 }
@@ -164,6 +172,8 @@ export interface PreparedLlmCall {
   readonly context?: LlmModelContext
   /** Exact model modalities captured with the adapter dispatch generation. */
   readonly inputModalities?: readonly ModelModality[]
+  /** Detached request-image budget the route enforces, when it declares one. */
+  readonly imageRequest?: LlmImageRequestBudget
   /** Config fields materialized by the captured adapter rather than proposed by the caller. */
   readonly adapterDefaults: LlmCallConfigAdapterDefaults
   /**
@@ -665,6 +675,33 @@ export class LlmRuntime extends TypertRemoteService {
     return modalities === undefined ? undefined : [...modalities]
   }
 
+  /** Validate and detach one adapter-declared request-image budget. */
+  private detachedImageRequest(
+    budget: LlmImageRequestBudget | undefined,
+    provider: string,
+    model: string,
+  ): LlmImageRequestBudget | undefined {
+    if (budget === undefined) return undefined
+    const positive = (value: number | undefined): boolean => value === undefined
+      || (Number.isSafeInteger(value) && value > 0)
+    if (!positive(budget.maxBytes) || !positive(budget.maxImages)
+      || !positive(budget.byteQuantum) || !positive(budget.countQuantum)
+      || !positive(budget.versionMaxBytes)) {
+      throw new LlmError(
+        `adapter returned invalid request-image budget for provider "${provider}" model "${model}"`,
+        'INVALID_MODEL_IMAGE_BUDGET',
+      )
+    }
+    return {
+      representation: budget.representation,
+      ...budget.maxBytes === undefined ? {} : { maxBytes: budget.maxBytes },
+      ...budget.maxImages === undefined ? {} : { maxImages: budget.maxImages },
+      ...budget.byteQuantum === undefined ? {} : { byteQuantum: budget.byteQuantum },
+      ...budget.countQuantum === undefined ? {} : { countQuantum: budget.countQuantum },
+      ...budget.versionMaxBytes === undefined ? {} : { versionMaxBytes: budget.versionMaxBytes },
+    }
+  }
+
   /**
    * Discover models advertised by one registered provider. Catalog membership
    * is advisory and never changes routing or request validation.
@@ -765,6 +802,7 @@ export class LlmRuntime extends TypertRemoteService {
         'INVALID_MODEL_MAX_TOKENS',
       )
     }
+    const imageRequest = this.detachedImageRequest(resolved.imageRequest, provider, model)
     const info: LlmResolvedModelInfo = {
       provider,
       id: model,
@@ -773,6 +811,7 @@ export class LlmRuntime extends TypertRemoteService {
       ...inputModalities === undefined ? {} : { inputModalities },
       ...context === undefined ? {} : { context: { contextWindow: context.contextWindow } },
       ...defaultMaxTokens === undefined ? {} : { defaultMaxTokens },
+      ...imageRequest === undefined ? {} : { imageRequest },
     }
     const reasoning = resolved.reasoning
     if (reasoning === undefined) return info
@@ -913,6 +952,9 @@ export class LlmRuntime extends TypertRemoteService {
       ...modelInfo.inputModalities === undefined
         ? {}
         : { inputModalities: Object.freeze([...modelInfo.inputModalities]) },
+      ...modelInfo.imageRequest === undefined
+        ? {}
+        : { imageRequest: deepFreeze(structuredClone(modelInfo.imageRequest)) },
       stream: (options: GenerateOptions): AsyncIterable<StreamChunk> => {
         if (dispatched) {
           throw new LlmError('a prepared LLM call can only be dispatched once', 'INVALID_PREPARED_CALL')

+ 42 - 2
packages/llm/llm/src/types.ts

@@ -48,6 +48,13 @@ export interface LlmFailure {
   readonly providerRetryAfterMs?: number
   /** Opaque provider-issued request identifier for diagnostics. */
   readonly requestId?: ProviderRequestId
+  /**
+   * With code `IMAGE_OFFLOAD_REQUIRED`: how many more of the oldest retained
+   * image occurrences the route needs offloaded before the same request fits
+   * its exact byte accounting. The agent loop advances the durable watermark
+   * by this count and rebuilds the request.
+   */
+  readonly offloadImages?: number
 }
 
 /** Plain text visible to the end user. */
@@ -72,6 +79,12 @@ export interface ImageBlock {
   type: 'image'
   /** Immutable bytes and intrinsic display metadata owned by the attachment service. */
   attachment: ImageAttachmentRef
+  /**
+   * Set by surface derivation when the occurrence lies at or before the
+   * session's durable `image/offload` watermark: every route sends its
+   * placeholder text instead of the image. Never stored in a logged message.
+   */
+  offloaded?: true
 }
 
 /** A tool invocation requested by the model. */
@@ -172,10 +185,11 @@ export interface LlmImageRequestPrice {
 export interface LlmImageRequestPricing {
   /**
    * Price every image occurrence of one request projection.
-   * @param images - durable image references in request order, one entry per occurrence.
+   * @param images - surface image blocks in request order, one entry per occurrence; an `offloaded` block
+   *   is priced as its placeholder text.
    * @returns one price per occurrence, aligned by index with `images`.
    */
-  priceImages(images: readonly ImageAttachmentRef[]): readonly LlmImageRequestPrice[]
+  priceImages(images: readonly ImageBlock[]): readonly LlmImageRequestPrice[]
 }
 
 /** Display metadata for one registered provider route. */
@@ -301,6 +315,30 @@ export interface LlmModelContext {
   contextWindow: number
 }
 
+/**
+ * Request-image budget one exact image-capable route declares, so the agent
+ * loop can advance the session's durable `image/offload` watermark before
+ * dispatch from logged facts alone. Byte accounting clamps each occurrence's
+ * normalized byte count to {@link versionMaxBytes} and, for the `base64`
+ * representation, expands it to its encoded length. A route whose exact
+ * request accounting still exceeds the budget fails the request with
+ * `IMAGE_OFFLOAD_REQUIRED` naming the additional occurrences to offload.
+ */
+export interface LlmImageRequestBudget {
+  /** Whether the route accounts raw file bytes or inline base64 length. */
+  representation: 'raw' | 'base64'
+  /** Accumulated represented image bytes the route accepts; absent leaves bytes unbounded. */
+  maxBytes?: number
+  /** Image occurrences the route accepts; absent leaves the count unbounded. */
+  maxImages?: number
+  /** Represented bytes removed as one deterministic advance step; absent removes the minimum. */
+  byteQuantum?: number
+  /** Occurrences removed as one deterministic advance step; absent removes the minimum. */
+  countQuantum?: number
+  /** Encoded-byte target of the route's derived request version; absent accounts normalized bytes. */
+  versionMaxBytes?: number
+}
+
 /** Display metadata for one adapter-owned reasoning effort. */
 export interface LlmReasoningEffortInfo {
   /** Opaque stable value accepted by {@link GenerateOptions.reasoningEffort}. */
@@ -330,6 +368,8 @@ export interface LlmResolvedModelInfo extends LlmModelInfo {
   defaultMaxTokens?: number
   /** Adapter-owned selectable reasoning levels when exposed. */
   reasoning?: LlmModelReasoningInfo
+  /** Request-image budget the route enforces; absent for routes that never offload. */
+  imageRequest?: LlmImageRequestBudget
 }
 
 /**

+ 100 - 114
packages/llm/llm/tests/content.spec.ts

@@ -3,30 +3,26 @@ import { AttachmentId, ImageVariantId } from '@deepseek-ai/dsh-attachment'
 import type { AttachmentStore, ImageMediaType } from '@deepseek-ai/dsh-attachment'
 import {
   ToolCallId,
+  compareImageBlockPaths,
   createUserMessage,
+  markOffloadedImages,
   offloadedImageText,
   offloadedImagePrefixCount,
-  offloadRequestImagesWithPolicy,
+  planImageOffload,
   projectImagesForTextModel,
+  projectOffloadedImages,
+  representedImageBytes,
   resolveImageAttachmentAccess,
   requestImageHandleText,
+  visitImageBlocks,
 } from '../src/index.ts'
-import type { ContentBlock, Message } from '../src/index.ts'
+import type { ContentBlock, ImageBlockPath } from '../src/index.ts'
 
 const source = { kind: 'plugin' as const, plugin: 'test' }
 
 const OMITTED = '[omitted]'
 
-function offloadBase64(messages: readonly Message[], maxBytes: number | undefined): readonly Message[] {
-  return offloadRequestImagesWithPolicy(messages, {
-    representation: 'base64',
-    ...maxBytes === undefined ? {} : { maxBytes },
-    byteQuantum: 1,
-    placeholder: () => OMITTED,
-  })
-}
-
-function image(bytes: number): Extract<ContentBlock, { type: 'image' }> {
+function image(bytes: number, offloaded?: true): Extract<ContentBlock, { type: 'image' }> {
   return {
     type: 'image',
     attachment: {
@@ -36,78 +32,109 @@ function image(bytes: number): Extract<ContentBlock, { type: 'image' }> {
       width: 1,
       height: 1,
     },
+    ...offloaded === undefined ? {} : { offloaded },
   }
 }
 
-describe('base64 request-image offload', () => {
-  it('preserves every image when no payload bound is configured', () => {
-    const messages = [createUserMessage({ content: [image(300)], source })]
-    expect(offloadBase64(messages, undefined)).toBe(messages)
+describe('visitImageBlocks', () => {
+  it('visits top-level and nested tool-result occurrences with their block paths', () => {
+    const seen: ImageBlockPath[] = []
+    visitImageBlocks([
+      { type: 'text', text: 'before' },
+      image(1),
+      { type: 'tool-result', toolCallId: ToolCallId('shot'), content: [{ type: 'text', text: 'x' }, image(2)] },
+      image(3),
+    ], (block, path) => seen.push([...path, block.attachment.bytes]))
+    expect(seen).toEqual([[1, 1], [2, 1, 2], [3, 3]])
   })
+})
 
-  it('preserves the original request when its base64 payload fits exactly', () => {
-    const messages = [createUserMessage({ content: [image(3), image(3)], source })]
-    expect(offloadBase64(messages, 8)).toBe(messages)
+describe('compareImageBlockPaths', () => {
+  it('orders paths lexicographically in message order', () => {
+    expect(compareImageBlockPaths([1], [2])).toBeLessThan(0)
+    expect(compareImageBlockPaths([2, 0], [2])).toBeGreaterThan(0)
+    expect(compareImageBlockPaths([2, 1], [2, 1])).toBe(0)
+    expect(compareImageBlockPaths([2, 1], [2, 3])).toBeLessThan(0)
   })
+})
 
-  it('keeps five 3 MiB images at 20 MiB and offloads the oldest after one more raw byte', () => {
-    const rawImageBytes = 3 * 1024 * 1024
-    const maxRequestImageBytes = 20 * 1024 * 1024
-    const exact = [createUserMessage({
-      content: Array.from({ length: 5 }, () => image(rawImageBytes)),
-      source,
-    })]
-    expect(offloadBase64(exact, maxRequestImageBytes)).toBe(exact)
+describe('representedImageBytes', () => {
+  it('clamps to the request-version target and expands inline representations', () => {
+    expect(representedImageBytes(100, { representation: 'raw' })).toBe(100)
+    expect(representedImageBytes(100, { representation: 'raw', versionMaxBytes: 60 })).toBe(60)
+    expect(representedImageBytes(3, { representation: 'base64' })).toBe(4)
+    expect(representedImageBytes(4, { representation: 'base64', versionMaxBytes: 60 })).toBe(8)
+  })
+})
 
-    const over = [createUserMessage({
-      content: [image(rawImageBytes + 1), ...Array.from({ length: 4 }, () => image(rawImageBytes))],
-      source,
-    })]
-    expect(offloadBase64(over, maxRequestImageBytes)[0]?.content).toEqual([
-      { type: 'text', text: OMITTED },
-      ...Array.from({ length: 4 }, () => image(rawImageBytes)),
+describe('markOffloadedImages', () => {
+  it('returns the original content when nothing lies at or before the watermark', () => {
+    const content: ContentBlock[] = [
+      { type: 'text', text: 'before' },
+      image(1),
+      { type: 'tool-result', toolCallId: ToolCallId('shot'), content: [image(2)] },
+    ]
+    expect(markOffloadedImages(content, () => false)).toBe(content)
+  })
+
+  it('marks matching top-level and nested occurrences without mutating durable content', () => {
+    const nested = image(2)
+    const content: ContentBlock[] = [
+      image(1),
+      { type: 'text', text: 'between' },
+      { type: 'tool-result', toolCallId: ToolCallId('shot'), content: [nested, image(3)] },
+      image(4),
+    ]
+    const marked = markOffloadedImages(content, path => compareImageBlockPaths(path, [2, 0]) <= 0)
+    expect(marked).toEqual([
+      image(1, true),
+      { type: 'text', text: 'between' },
+      { type: 'tool-result', toolCallId: ToolCallId('shot'), content: [image(2, true), image(3)] },
+      image(4),
     ])
+    expect(content[0]).toEqual(image(1))
+    expect(nested).toEqual(image(2))
   })
 
-  it('replaces the oldest nested occurrences without mutating durable messages', () => {
-    const shared = image(3)
+  it('leaves an already marked occurrence untouched', () => {
+    const content = [image(1, true)]
+    expect(markOffloadedImages(content, () => true)).toBe(content)
+  })
+})
+
+describe('projectOffloadedImages', () => {
+  it('keeps messages without offloaded occurrences by identity', () => {
+    const messages = [createUserMessage({ content: [image(300)], source })]
+    const projected = projectOffloadedImages(messages, () => OMITTED)
+    expect(projected[0]).toBe(messages[0])
+  })
+
+  it('replaces marked top-level and nested occurrences with route placeholders', () => {
     const messages = [
       createUserMessage({
-        content: [{
-          type: 'tool-result',
-          toolCallId: ToolCallId('shot'),
-          content: [shared],
-        }],
+        content: [{ type: 'tool-result', toolCallId: ToolCallId('shot'), content: [image(3, true)] }],
         source,
       }),
-      createUserMessage({ content: [shared, image(3)], source }),
+      createUserMessage({ content: [image(3, true), image(3)], source }),
     ]
-
-    const fitted = offloadBase64(messages, 8)
-    expect(fitted).not.toBe(messages)
-    expect(fitted[0]?.content).toEqual([{
+    const projected = projectOffloadedImages(messages, ref => `${OMITTED}:${ref.bytes}`)
+    expect(projected[0]?.content).toEqual([{
       type: 'tool-result',
       toolCallId: ToolCallId('shot'),
-      content: [{ type: 'text', text: OMITTED }],
+      content: [{ type: 'text', text: `${OMITTED}:3` }],
     }])
-    expect(fitted[1]?.content).toEqual([shared, image(3)])
-    expect(messages[0]?.content[0]).toMatchObject({ type: 'tool-result', content: [shared] })
-  })
-
-  it('replaces a single image that cannot fit', () => {
-    const messages = [createUserMessage({ content: [image(300)], source })]
-    expect(offloadBase64(messages, 8)[0]?.content)
-      .toEqual([{ type: 'text', text: OMITTED }])
+    expect(projected[1]?.content).toEqual([{ type: 'text', text: `${OMITTED}:3` }, image(3)])
+    expect(messages[1]?.content[0]).toEqual(image(3, true))
   })
 
-  it('keeps unchanged nested content while replacing a later image', () => {
+  it('keeps unchanged nested content while replacing a later occurrence', () => {
     const nested = {
       type: 'tool-result' as const,
       toolCallId: ToolCallId('text-only'),
       content: [{ type: 'text' as const, text: 'kept' }],
     }
-    const messages = [createUserMessage({ content: [nested, image(3)], source })]
-    expect(offloadBase64(messages, 1)[0]?.content).toEqual([
+    const messages = [createUserMessage({ content: [nested, image(3, true)], source })]
+    expect(projectOffloadedImages(messages, () => OMITTED)[0]?.content).toEqual([
       nested,
       { type: 'text', text: OMITTED },
     ])
@@ -127,67 +154,26 @@ describe('offloadedImagePrefixCount', () => {
   })
 })
 
-describe('offloadRequestImagesWithPolicy', () => {
-  it('drops 129 MiB to 64 MiB and keeps the removed prefix stable through 192 MiB', () => {
-    const mib = 1024 * 1024
-    const project = (count: number) => offloadRequestImagesWithPolicy([
-      createUserMessage({ content: Array.from({ length: count }, () => image(mib)), source }),
-    ], {
-      representation: 'raw',
-      maxBytes: 128 * mib,
-      byteQuantum: 64 * mib,
-      placeholder: () => OMITTED,
-    })[0]?.content
+describe('planImageOffload', () => {
+  const mib = 1024 * 1024
+  const retained = (count: number) => Array.from({ length: count }, (_, index) => ({ position: index, bytes: mib }))
 
-    expect(project(128)?.filter(block => block.type === 'image')).toHaveLength(128)
-    expect(project(129)?.filter(block => block.type === 'text')).toHaveLength(65)
-    expect(project(192)?.filter(block => block.type === 'text')).toHaveLength(65)
-    expect(project(193)?.filter(block => block.type === 'text')).toHaveLength(129)
+  it('drops 129 retained MiB to 64 MiB and holds until the retained total exceeds the bound again', () => {
+    const budget = { representation: 'raw' as const, maxBytes: 128 * mib, byteQuantum: 64 * mib }
+    expect(planImageOffload(retained(128), budget)).toBeUndefined()
+    expect(planImageOffload(retained(129), budget)).toBe(64)
+    expect(planImageOffload(retained(64), budget)).toBeUndefined()
   })
 
-  it('rounds a count excess up to a 20-image removal step', () => {
-    const projected = offloadRequestImagesWithPolicy([
-      createUserMessage({ content: Array.from({ length: 601 }, () => image(1)), source }),
-    ], {
-      representation: 'raw',
-      maxImages: 600,
-      countQuantum: 20,
-      placeholder: () => OMITTED,
-    })
-    expect(projected[0]?.content.filter(block => block.type === 'text')).toHaveLength(20)
-    expect(projected[0]?.content.filter(block => block.type === 'image')).toHaveLength(581)
-  })
-
-  it('uses route-owned request byte lengths when supplied', () => {
-    const messages = [createUserMessage({ content: [image(100), image(100)], source })]
-    const projected = offloadRequestImagesWithPolicy(messages, {
-      representation: 'raw',
-      maxBytes: 3,
-      byteLength: () => 2,
-      placeholder: () => OMITTED,
-    })
-    expect(projected[0]?.content).toEqual([
-      { type: 'text', text: OMITTED },
-      image(100),
-    ])
+  it('rounds a count excess up to a 20-image advance', () => {
+    const budget = { representation: 'raw' as const, maxImages: 600, countQuantum: 20 }
+    expect(planImageOffload(retained(601), budget)).toBe(19)
   })
 
-  it('builds a distinct placeholder from each omitted attachment', () => {
-    const first = image(3)
-    const second = image(3)
-    first.attachment = { ...first.attachment, name: 'first.png' }
-    second.attachment = { ...second.attachment, name: 'second.png' }
-    const projected = offloadRequestImagesWithPolicy([
-      createUserMessage({ content: [first, second], source }),
-    ], {
-      representation: 'raw',
-      maxBytes: 3,
-      placeholder: ref => `omitted:${ref.name}`,
-    })
-    expect(projected[0]?.content).toEqual([
-      { type: 'text', text: 'omitted:first.png' },
-      second,
-    ])
+  it('accounts inline representations by base64 length after the version clamp', () => {
+    const budget = { representation: 'base64' as const, maxBytes: 8, versionMaxBytes: 3 }
+    expect(planImageOffload([{ position: 'a', bytes: 3 }, { position: 'b', bytes: 3 }], budget)).toBeUndefined()
+    expect(planImageOffload([{ position: 'a', bytes: 300 }, { position: 'b', bytes: 3 }, { position: 'c', bytes: 3 }], budget)).toBe('a')
   })
 })
 

+ 30 - 0
packages/llm/llm/tests/service.spec.ts

@@ -1338,3 +1338,33 @@ describe('LlmRuntime', () => {
     expect(ctx.llm.listProviders()).toEqual([])
   })
 })
+
+describe('request-image budget metadata', () => {
+  it('detaches a valid budget onto the prepared call and rejects a non-positive field', async () => {
+    class BudgetAdapter extends LlmAdapter {
+      constructor(private readonly budget: Record<string, unknown>) {
+        super()
+      }
+
+      override resolveModel(provider: string, model: string) {
+        return Promise.resolve({ provider, id: model, name: model, imageRequest: this.budget as never })
+      }
+
+      async * stream(): AsyncIterable<never> {
+        throw new Error('unused')
+      }
+    }
+    const ctx = new Context()
+    const llm = new LlmRuntime(ctx)
+    const full = { representation: 'raw', maxBytes: 8, maxImages: 3, byteQuantum: 4, countQuantum: 2, versionMaxBytes: 5 }
+    llm.registerAdapter(['good'], new BudgetAdapter(full))
+    llm.registerAdapter(['bare'], new BudgetAdapter({ representation: 'base64' }))
+    llm.registerAdapter(['bad'], new BudgetAdapter({ representation: 'base64', maxImages: 0 }))
+    const prepared = await llm.prepareCall({ provider: 'good', model: 'm' })
+    expect(prepared.imageRequest).toEqual(full)
+    expect((await llm.prepareCall({ provider: 'bare', model: 'm' })).imageRequest).toEqual({ representation: 'base64' })
+    expect(Object.isFrozen(prepared.imageRequest)).toBe(true)
+    await expect(llm.prepareCall({ provider: 'bad', model: 'm' })).rejects.toMatchObject({ code: 'INVALID_MODEL_IMAGE_BUDGET' })
+    expect(() => new LlmError('x', 'y', { offloadImages: 0 })).toThrow('offloadImages must be a positive safe integer')
+  })
+})

+ 2 - 2
packages/llm/token-meter/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/llm/token-meter/README.md
-README.md: f2a4f7f7c70de4a08fd824e6e8001622ddb456b5
-README.zh.md: 3b163198279d4b254cd78638ebcad479d063630f
+README.md: 6871dc5097e2f1c493dfd4c883af4a0ceaa5f39a
+README.zh.md: cab6eb0c283d0797efd73409d30f5cf32ebef287

+ 1 - 1
packages/llm/token-meter/README.md

@@ -40,7 +40,7 @@ const { totalTokens, surfaceTokens, nodes } = ctx.tokenMeter.measure(session)
 const price = ctx.tokenMeter.estimateMessage(message)
 ```
 
-Each measurement resolves the effective envelope's provider/model through the optional `llm` service. Image occurrences use the routed request's visual-token price plus model-visible text when the adapter declares pricing; other routes keep the fixed heuristic. Each node also carries route-independent `heuristicTokens` for replacement shadow prices. Provider usage is reused only when the latest successful call's canonical request envelope matches the measured envelope and its total is no lower than that call's full route-priced anchor; otherwise the complete current envelope and surface are estimated. Surface changes stay signed relative to a matching anchor repriced under the same route, including negative deltas after shrinking replacements.
+Each measurement resolves the effective envelope's provider/model through the optional `llm` service. Image occurrences use the routed request's visual-token price plus model-visible text when the adapter declares pricing, and occurrences at or before the session's durable `image/offload` watermark are priced as the route's placeholder text exactly as the derived surface sends them; other routes keep the fixed heuristic. Each node also carries route-independent `heuristicTokens` for replacement shadow prices. Provider usage is reused only when the latest successful call's canonical request envelope matches the measured envelope and its total is no lower than that call's full route-priced anchor; otherwise the complete current envelope and surface are estimated. Surface changes stay signed relative to a matching anchor repriced under the same route, including negative deltas after shrinking replacements.
 
 ### Session projections
 

+ 1 - 1
packages/llm/token-meter/README.zh.md

@@ -40,7 +40,7 @@ const { totalTokens, surfaceTokens, nodes } = ctx.tokenMeter.measure(session)
 const price = ctx.tokenMeter.estimateMessage(message)
 ```
 
-每次测量都会通过可选的 `llm` 服务解析生效 envelope 的提供方/模型。适配器声明图片定价时,图片出现处使用路由请求的视觉 token 价格加模型可见文本;其他路由保持固定启发式规则。每个节点还携带与路由无关的 `heuristicTokens`,供替换影子价使用。只有当最新成功调用的规范请求 envelope 与已测量 envelope 匹配、且其总量不低于该调用完整路由定价锚点时,才复用提供方用量;否则会对完整当前 envelope 与表面做估算。表面变更保持相对于按同一路由重新定价的匹配锚点的带符号值,包括缩减替换后的负 delta。
+每次测量都会通过可选的 `llm` 服务解析生效 envelope 的提供方/模型。适配器声明图片定价时,图片出现处使用路由请求的视觉 token 价格加模型可见文本,位于会话持久 `image/offload` 水位及之前的出现位置则按路由的占位文本计价,与派生表层实际发送的一致;其他路由保持固定启发式规则。每个节点还携带与路由无关的 `heuristicTokens`,供替换影子价使用。只有当最新成功调用的规范请求 envelope 与已测量 envelope 匹配、且其总量不低于该调用完整路由定价锚点时,才复用提供方用量;否则会对完整当前 envelope 与表面做估算。表面变更保持相对于按同一路由重新定价的匹配锚点的带符号值,包括缩减替换后的负 delta。
 
 ### 会话投影
 

+ 15 - 2
packages/llm/token-meter/src/index.ts

@@ -10,6 +10,7 @@ import { BlockAssembler } from '@deepseek-ai/dsh-llm'
 import type { LlmImageRequestPricing, Message, TokenUsage } from '@deepseek-ai/dsh-llm'
 import { deepFreeze } from '@deepseek-ai/dsh-util-values'
 import type {
+  ImageOccurrencePosition,
   EpochHeader,
   Session,
   SessionEvent,
@@ -47,6 +48,8 @@ interface MeasurementAnchor {
   readonly header: EpochHeader | undefined
   /** Surface snapshot the anchored request was derived from. */
   readonly nodes: readonly MeterSurfaceNode[]
+  /** Offload watermark the anchored request was derived under. */
+  readonly watermark: ImageOccurrencePosition | undefined
   /** Fixed-heuristic price of the call's provider output. */
   readonly assistantTokens: number
   /** Provider usage of the call, when it reported one under a known header. */
@@ -57,6 +60,8 @@ interface ReplayState {
   consumedEvents: SessionLogOffsetType
   header: EpochHeader | undefined
   surface: MeterSurfaceNode[]
+  /** Durable `image/offload` watermark in force after the consumed events. */
+  watermark: ImageOccurrencePosition | undefined
   stepStart: { turn: number; step: number; nodes: readonly MeterSurfaceNode[] } | undefined
   anchor: MeasurementAnchor | undefined
 }
@@ -142,7 +147,7 @@ export class TokenMeter extends Service {
       ? state.header
       : canonicalHeader(requestHeader)
     const pricing = this._routeImagePricing(header)
-    const surface = priceSurface(state.surface, pricing)
+    const surface = priceSurface(state.surface, pricing, state.watermark)
     const anchor = state.anchor
 
     let baseline: TokenMeasurementBaseline
@@ -151,7 +156,7 @@ export class TokenMeter extends Service {
       // Matching headers share one route, so the anchored snapshot reprices
       // under the same pricing as the current surface and the signed delta
       // compares like with like.
-      const anchorSurfaceTokens = priceSurface(anchor.nodes, pricing).surfaceTokens
+      const anchorSurfaceTokens = priceSurface(anchor.nodes, pricing, anchor.watermark).surfaceTokens
         + anchor.assistantTokens
       const estimatedAnchorTokens = estimateHeader(header) + anchorSurfaceTokens
       const usage = anchor.usage
@@ -207,6 +212,7 @@ export class TokenMeter extends Service {
         consumedEvents: SessionLogOffset(0),
         header: undefined,
         surface: [],
+        watermark: undefined,
         stepStart: undefined,
         anchor: undefined,
       }
@@ -229,6 +235,7 @@ export class TokenMeter extends Service {
    */
   private _foldEvent(session: Session, state: ReplayState, event: SessionEvent): void {
     let nextHeader = state.header
+    let nextWatermark = state.watermark
     let nextStepStart = state.stepStart
     let nextAnchor = state.anchor
 
@@ -236,6 +243,9 @@ export class TokenMeter extends Service {
       case 'request/header':
         nextHeader = canonicalHeader(event.data.header)
         break
+      case 'image/offload':
+        nextWatermark = event.data.watermark
+        break
       case 'step/start':
         if (state.stepStart !== undefined) {
           throw new Error(
@@ -275,6 +285,7 @@ export class TokenMeter extends Service {
         nextAnchor = {
           header: nextHeader,
           nodes: stepStart.nodes,
+          watermark: nextWatermark,
           assistantTokens: this._estimateProviderAssistant(session, event, eventTokens),
           usage: event.data.usage,
         }
@@ -282,6 +293,7 @@ export class TokenMeter extends Service {
         nextAnchor = {
           header: nextHeader,
           nodes: stepStart.nodes,
+          watermark: nextWatermark,
           assistantTokens: eventTokens,
           usage: undefined,
         }
@@ -289,6 +301,7 @@ export class TokenMeter extends Service {
     }
 
     state.header = nextHeader
+    state.watermark = nextWatermark
     state.stepStart = nextStepStart
     if (plan !== undefined) {
       commitSurfaceTokens(state.surface, plan)

+ 15 - 3
packages/llm/token-meter/src/route-pricing.ts

@@ -8,7 +8,9 @@
  * @module @deepseek-ai/dsh-token-meter/route-pricing
  */
 
-import type { LlmImageRequestPricing } from '@deepseek-ai/dsh-llm'
+import type { ImageBlock, LlmImageRequestPricing } from '@deepseek-ai/dsh-llm'
+import { compareImagePositions } from '@deepseek-ai/dsh-session'
+import type { ImageOccurrencePosition } from '@deepseek-ai/dsh-session'
 import { estimateContent } from './estimate.ts'
 import type { MeterSurfaceNode } from './surface-fold.ts'
 import type { TokenSurfaceNode } from './types.ts'
@@ -22,9 +24,12 @@ export interface PricedSurface {
 }
 
 /**
- * Price one ordered surface under a route's request-image pricing.
+ * Price one ordered surface under a route's request-image pricing. Every
+ * occurrence positioned at or before `watermark` is priced as the route's
+ * offloaded placeholder, exactly as the derived surface sends it.
  * @param nodes - the fold's current or snapshotted surface, in model-visible order.
  * @param pricing - the routed model's image pricing, or undefined to keep the fixed heuristic.
+ * @param watermark - the durable `image/offload` watermark in force for this surface, if any.
  * @returns detached public nodes and their route-priced total.
  * @throws when the pricing answers a different occurrence count than it was
  *   asked — misalignment would silently misprice nodes, so it must fail loud.
@@ -32,8 +37,15 @@ export interface PricedSurface {
 export function priceSurface(
   nodes: readonly MeterSurfaceNode[],
   pricing: LlmImageRequestPricing | undefined,
+  watermark?: ImageOccurrencePosition,
 ): PricedSurface {
-  const images = pricing === undefined ? [] : nodes.flatMap(node => node.images)
+  const images: ImageBlock[] = pricing === undefined
+    ? []
+    : nodes.flatMap(node => node.images.map(({ attachment, path }) => (
+      watermark !== undefined && compareImagePositions({ seq: node.seq, path: [...path] }, watermark) <= 0
+        ? { type: 'image' as const, attachment, offloaded: true as const }
+        : { type: 'image' as const, attachment }
+    )))
   if (pricing === undefined || images.length === 0) {
     let surfaceTokens = 0
     const publicNodes = nodes.map((node) => {

+ 18 - 13
packages/llm/token-meter/src/surface-fold.ts

@@ -18,10 +18,19 @@
 
 import { deriveEventMessage } from '@deepseek-ai/dsh-session'
 import type { SessionSeq, SurfaceEvent } from '@deepseek-ai/dsh-session'
-import type { ContentBlock, Message } from '@deepseek-ai/dsh-llm'
+import { visitImageBlocks } from '@deepseek-ai/dsh-llm'
+import type { ContentBlock, ImageBlockPath, Message } from '@deepseek-ai/dsh-llm'
 import type { ImageAttachmentRef } from '@deepseek-ai/dsh-attachment'
 import { estimateMessage, estimateStructuralBlock } from './estimate.ts'
 
+/** One durable image occurrence of a surface node, positioned for the offload watermark. */
+export interface MeterImageOccurrence {
+  /** Durable normalized attachment reference of the occurrence. */
+  readonly attachment: ImageAttachmentRef
+  /** Block path of the occurrence inside the node's message content. */
+  readonly path: ImageBlockPath
+}
+
 /** One priced surface node with the image occurrences route pricing replaces. */
 export interface MeterSurfaceNode {
   /** Durable sequence number of the surface event. */
@@ -31,7 +40,7 @@ export interface MeterSurfaceNode {
   /** Fixed-heuristic price with every image occurrence's structural price removed. */
   readonly imageFreeTokens: number
   /** Durable image occurrences in message order; empty for image-free nodes. */
-  readonly images: readonly ImageAttachmentRef[]
+  readonly images: readonly MeterImageOccurrence[]
 }
 
 /** One validated surface transition that has not mutated the priced surface yet. */
@@ -46,17 +55,13 @@ export interface SurfaceTokenPlan {
   readonly target: 'append' | { readonly startIdx: number; readonly endIdx: number }
 }
 
-/** Collect image occurrences recursively and total their structural prices. */
-function collectImages(blocks: readonly ContentBlock[], images: ImageAttachmentRef[]): number {
+/** Collect image occurrences with their block paths and total their structural prices. */
+function collectImages(blocks: readonly ContentBlock[], images: MeterImageOccurrence[]): number {
   let structuralTokens = 0
-  for (const block of blocks) {
-    if (block.type === 'image') {
-      images.push(block.attachment)
-      structuralTokens += estimateStructuralBlock(block)
-    } else if (block.type === 'tool-result') {
-      structuralTokens += collectImages(block.content, images)
-    }
-  }
+  visitImageBlocks(blocks, (block, path) => {
+    images.push({ attachment: block.attachment, path })
+    structuralTokens += estimateStructuralBlock(block)
+  })
   return structuralTokens
 }
 
@@ -64,7 +69,7 @@ function collectImages(blocks: readonly ContentBlock[], images: ImageAttachmentR
 function analyzeNode(seq: SessionSeq, message: Message | null): MeterSurfaceNode {
   if (message === null) return { seq, heuristicTokens: 0, imageFreeTokens: 0, images: [] }
   const heuristicTokens = estimateMessage(message)
-  const images: ImageAttachmentRef[] = []
+  const images: MeterImageOccurrence[] = []
   const imageStructuralTokens = collectImages(message.content, images)
   return {
     seq,

+ 26 - 0
packages/llm/token-meter/tests/route-pricing.spec.ts

@@ -165,6 +165,32 @@ describe('route-aware image pricing', () => {
     expect(unknownRoute.nodes[0]!.tokens).toBe(estimateMessage(message))
   })
 
+  it('prices occurrences at or before the image/offload watermark as the route placeholder', async () => {
+    const placeholder = '[offloaded]'
+    const watermarkPricing: LlmImageRequestPricing = {
+      priceImages: images => images.map(block => (block.offloaded === true
+        ? { visualTokens: 0, text: placeholder }
+        : { visualTokens: VISUAL_TOKENS, text: HANDLE_TEXT })),
+    }
+    const { meter, session } = await harness(() => watermarkPricing)
+    session.append('turn/start', { turn: 1 })
+    const older = imageMessage('older')
+    const newer = imageMessage('newer')
+    const olderSeq = session.append('user/message', older, { surfaceOp: 'append' }).seq
+    session.append('user/message', newer, { surfaceOp: 'append' })
+    session.append('request/header', { header: header('vision'), reason: 'initial' })
+    const before = meter.measure(session)
+    expect(before.nodes.map(node => node.tokens)).toEqual([routedMessageTokens(older), routedMessageTokens(newer)])
+
+    session.append('step/start', { turn: 1, step: 1 })
+    session.append('image/offload', { turn: 1, step: 1, watermark: { seq: olderSeq, path: [1] } })
+    const after = meter.measure(session)
+    const imageFree = estimateMessage({ ...older, content: older.content.filter(block => block.type !== 'image') })
+    expect(after.nodes[0]!.tokens).toBe(imageFree + estimateContent([{ type: 'text', text: placeholder }]))
+    expect(after.nodes[1]!.tokens).toBe(routedMessageTokens(newer))
+    expect(after.totalTokens).toBeLessThan(before.totalTokens)
+  })
+
   it('fails loud when a route answers a mismatched occurrence count', async () => {
     const broken: LlmImageRequestPricing = { priceImages: () => [] }
     const { meter, session } = await harness(() => broken)

+ 2 - 2
packages/test-support/llm-replay/README.i18n.yaml

@@ -2,5 +2,5 @@
 # side as of the last confirmed-consistent state. Both languages carry equal authority;
 # after editing either side, bring the other along and re-record with:
 #   pnpm run verify-translation-pairing --write packages/test-support/llm-replay/README.md
-README.md: c0643bf6823e96cf3663f8299d0f5645d7974011
-README.zh.md: 1c2fe3393691b4988678497aa5948957f678c80b
+README.md: edb05c3356aa4f438a158cbddcad7913adc85bf3
+README.zh.md: 7688093b598509fb35cb6987b340eb3891361f57

+ 1 - 1
packages/test-support/llm-replay/README.md

@@ -58,7 +58,7 @@ With `providers` configured, the plugin registers a replay-only adapter whose ca
 | `file` | `$DSH_SNAPSHOT_FILE` | Path to the primary (parent) `session.jsonl` fixture; required (config or env) |
 | `overrideFile` | `$DSH_SNAPSHOT_OVERRIDE` | Optional `ReplayOverrideDoc` sidecar for the primary session |
 | `childFiles` | `$DSH_SNAPSHOT_CHILD_FILES` | Recorded subagent child-session logs for a nested scenario |
-| `providers` | — | Optional replay-only provider and model catalog; a model may declare `contextWindow`, text/image modalities, and positive `imageRequestTokens` when image-capable; invalid values fail at load and routes never perform provider I/O |
+| `providers` | — | Optional replay-only provider and model catalog; a model may declare `contextWindow`, text/image modalities, and, when image-capable, positive `imageRequestTokens` and a positive `imageRequestMaxBytes` base64 budget that drives the agent loop's `image/offload` watermark; invalid values fail at load and routes never perform provider I/O |
 | `paceMs` | — (burst) | Optional per-chunk delay in ms for genuinely incremental delivery |
 
 The generated [configuration catalog](../../../docs/config-catalog.md#deepseek-aidsh-llm-replay) is the exhaustive source for every accepted field and its JSDoc.

+ 1 - 1
packages/test-support/llm-replay/README.zh.md

@@ -58,7 +58,7 @@ kind: "package-reference"
 | `file` | `$DSH_SNAPSHOT_FILE` | 主(父)`session.jsonl` fixture 的路径;必需(配置或 env) |
 | `overrideFile` | `$DSH_SNAPSHOT_OVERRIDE` | 主会话的可选 `ReplayOverrideDoc` 伴随文件 |
 | `childFiles` | `$DSH_SNAPSHOT_CHILD_FILES` | 嵌套场景中已记录的 subagent 子会话日志 |
-| `providers` | 无 | 可选的仅回放提供方与模型目录;模型可声明 `contextWindow`、文本/图片模态,以及图片模型使用的正整数 `imageRequestTokens`;非法值会在加载时失败,路由绝不执行提供方 I/O |
+| `providers` | 无 | 可选的仅回放提供方与模型目录;模型可声明 `contextWindow`、文本/图片模态,以及图片模型使用的正整数 `imageRequestTokens` 和驱动 agent loop `image/offload` 水位的正整数 base64 预算 `imageRequestMaxBytes`;非法值会在加载时失败,路由绝不执行提供方 I/O |
 | `paceMs` | 无(突发) | 可选的每分片延迟(毫秒),用于真正的增量投递 |
 
 生成的[配置目录](../../../docs/config-catalog.zh.md#deepseek-aidsh-llm-replay)是每个受支持字段及其 JSDoc 的穷尽式真源。

+ 28 - 5
packages/test-support/llm-replay/src/index.ts

@@ -27,7 +27,7 @@ import type {
   StreamChunk,
   TokenUsage,
 } from '@deepseek-ai/dsh-llm'
-import { LlmAdapter, LlmError, ReasoningEffortId, requestImageHandleText, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
+import { LlmAdapter, LlmError, ReasoningEffortId, offloadedImageText, requestImageHandleText, resolveRetryPolicy } from '@deepseek-ai/dsh-llm'
 import { assertNever } from '@deepseek-ai/dsh-util-values'
 
 const PACKED_CHUNK_ROW_TYPES = new Set(['text-chunks', 'reasoning-chunks', 'tool-call-chunks'])
@@ -73,6 +73,13 @@ export interface ReplayModelConfig {
    * no image pricing.
    */
   imageRequestTokens?: number
+  /**
+   * Optional accumulated base64 image-byte bound the replay route declares,
+   * so the agent loop advances the durable `image/offload` watermark in
+   * keyless scenarios exactly as a live image-capable route would. Requires
+   * {@link inputModalities} to include `image`. Absent declares no budget.
+   */
+  imageRequestMaxBytes?: number
   /** Optional reasoning-effort ids the replay route accepts, in display order. */
   reasoningEfforts?: string[]
   /**
@@ -683,10 +690,9 @@ class ReplayAdapter extends LlmAdapter {
     const visualTokens = configured?.models?.find(candidate => candidate.id === model)?.imageRequestTokens
     if (visualTokens === undefined) return undefined
     return {
-      priceImages: images => images.map(ref => ({
-        visualTokens,
-        text: requestImageHandleText(ref, { width: ref.width, height: ref.height }),
-      })),
+      priceImages: images => images.map(({ attachment: ref, offloaded }) => (offloaded === true
+        ? { visualTokens: 0, text: offloadedImageText(ref) }
+        : { visualTokens, text: requestImageHandleText(ref, { width: ref.width, height: ref.height }) })),
     }
   }
 
@@ -722,6 +728,9 @@ class ReplayAdapter extends LlmAdapter {
       ...configuredModel?.defaultMaxTokens === undefined
         ? {}
         : { defaultMaxTokens: configuredModel.defaultMaxTokens },
+      ...configuredModel?.imageRequestMaxBytes === undefined
+        ? {}
+        : { imageRequest: { representation: 'base64' as const, maxBytes: configuredModel.imageRequestMaxBytes } },
       ...configuredModel?.reasoningEfforts === undefined
         ? {}
         : {
@@ -966,6 +975,20 @@ function validateConfiguredModels(providers: ReplayProviderConfig[] | undefined)
           + 'requires inputModalities to include "image"',
         )
       }
+      const imageRequestMaxBytes: unknown = model.imageRequestMaxBytes
+      if (imageRequestMaxBytes !== undefined
+        && (!Number.isSafeInteger(imageRequestMaxBytes) || (imageRequestMaxBytes as number) <= 0)) {
+        throw new Error(
+          `llm-replay: provider "${provider.id}" model "${model.id}" imageRequestMaxBytes `
+          + 'must be a positive safe integer',
+        )
+      }
+      if (imageRequestMaxBytes !== undefined && model.inputModalities?.includes('image') !== true) {
+        throw new Error(
+          `llm-replay: provider "${provider.id}" model "${model.id}" imageRequestMaxBytes `
+          + 'requires inputModalities to include "image"',
+        )
+      }
     }
   }
 }

+ 38 - 3
packages/test-support/llm-replay/tests/llm-replay.spec.ts

@@ -1380,12 +1380,47 @@ describe('apply (the plugin entry)', () => {
       width: 640,
       height: 480,
     } as never
-    const priced = pricing?.priceImages([ref, ref])
-    expect(priced?.map(price => price.visualTokens)).toEqual([384, 384])
-    expect(priced?.every(price => price.text.includes('640x480px'))).toBe(true)
+    const priced = pricing?.priceImages([
+      { type: 'image', attachment: ref },
+      { type: 'image', attachment: ref, offloaded: true },
+    ])
+    expect(priced?.map(price => price.visualTokens)).toEqual([384, 0])
+    expect(priced?.[0]?.text).toContain('640x480px')
+    expect(priced?.[1]?.text).toContain('image omitted to fit request image limits')
     expect(ctx.llm.imageRequestPricing('deepseek', 'plain')).toBeUndefined()
   })
 
+  it('declares a base64 request-image budget only for models that configure it', async () => {
+    writeFileSync(file, sessionJsonl(TEXT_CHUNKS.map((c, i) => chunkEvent(SessionSeq(i + 1), 1, 1, c))), 'utf8')
+    const ctx = new Context()
+    await ctx.plugin(LlmRuntime)
+    installLlmReplay(ctx, {
+      file,
+      providers: [{
+        id: 'deepseek',
+        models: [
+          { id: 'vision', inputModalities: ['text', 'image'], imageRequestMaxBytes: 4096 },
+          { id: 'plain' },
+        ],
+      }],
+    })
+    await expect(ctx.llm.resolveModelInfo('deepseek', 'vision')).resolves.toMatchObject({
+      imageRequest: { representation: 'base64', maxBytes: 4096 },
+    })
+    expect((await ctx.llm.resolveModelInfo('deepseek', 'plain')).imageRequest).toBeUndefined()
+  })
+
+  it.each([
+    [{ id: 'm', inputModalities: ['text', 'image'], imageRequestMaxBytes: 0 }, 'must be a positive safe integer'],
+    [{ id: 'm', imageRequestMaxBytes: 4096 }, 'requires inputModalities to include "image"'],
+  ])('rejects an invalid imageRequestMaxBytes during load: %j', (model, reason) => {
+    const ctx = new Context()
+    const providers = [{ id: 'm', models: [model] }] as unknown as NonNullable<Config['providers']>
+    expect(() => { apply(ctx, { file, providers }) }).toThrow(
+      `llm-replay: provider "m" model "m" imageRequestMaxBytes ${reason}`,
+    )
+  })
+
   it('rejects imageRequestTokens on a model without the image modality during load', () => {
     const ctx = new Context()
     const providers = [{ id: 'm', models: [{ id: 'm', imageRequestTokens: 384 }] }] as unknown as

+ 5 - 0
scripts/type-equiv.manifest.json

@@ -469,6 +469,11 @@
       "symbol": "RequestContext",
       "source": "packages/core/session/src/types.ts"
     },
+    {
+      "doc": "docs/subsystems/session.md",
+      "symbol": "ImageOccurrencePosition",
+      "source": "packages/core/session/src/types.ts"
+    },
     {
       "doc": "docs/subsystems/session.md",
       "symbol": "SessionSeq",

+ 5 - 0
snapshots/acp/acp.snapshot.ts

@@ -40,6 +40,11 @@ const controllerCases: readonly {
     hasModelTurn: true,
     configPath: join(corpusDir, 'image-compaction', 'cordis.yml'),
   },
+  {
+    name: 'image-offload',
+    hasModelTurn: true,
+    configPath: join(corpusDir, 'image-offload', 'cordis.yml'),
+  },
 ] as const
 
 function localScenarioSource(source: string | undefined): string | undefined {

+ 56 - 0
snapshots/acp/image-offload/cordis.snapshot.yml

@@ -0,0 +1,56 @@
+# Keyless replay for the image-offload scenario. This profile patch swaps the
+# adapter and re-pins the recorded vision model. The replay catalog declares
+# image input and a base64 request-image budget that six 69-byte frames
+# (92 base64 bytes each) exceed, so the agent loop appends an `image/offload`
+# watermark before the first dispatch and the second turn sends placeholders.
+- id: llm-deepseek
+  name: '@deepseek-ai/dsh-llm-deepseek'
+  disabled: true
+
+- id: acp
+  name: '@deepseek-ai/dsh-acp'
+  config:
+    provider: deepseek-official
+    model: deepseek-v4-flash-vision-exp
+
+- id: session-persistence-jsonl
+  name: '@deepseek-ai/dsh-session-persistence-jsonl'
+  config:
+    root: !!js process.env.DSH_SNAPSHOT_SESSIONS_ROOT ?? './.sessions'
+    compression: none
+
+- id: agent-instructions
+  name: '@deepseek-ai/dsh-agent-instructions'
+  config:
+    maxBytes: 65536
+
+- id: system-prompt
+  name: '@deepseek-ai/dsh-system-prompt'
+  config:
+    persona: |
+      You are a coding assistant powered by the {{model}} model. Your working directory is {{cwd}}. Your bash tool runs under a file sandbox — a `[sandbox: file access denied …]` result is policy, not a command bug.
+
+      Verify your work by running the code or tests. Keep answers brief and factual.
+
+- insert:
+    - id: llm-replay
+      name: '@deepseek-ai/dsh-llm-replay'
+      config:
+        providers:
+          - id: deepseek-official
+            name: DeepSeek
+            models:
+              - id: deepseek-v4-flash
+                inputModalities: [text]
+              - id: deepseek-v4-pro
+                inputModalities: [text]
+              - id: deepseek-v4-flash-vision-exp
+                inputModalities: [text, image]
+                # A wide context window keeps automatic compaction out of the
+                # scenario; the base64 budget admits three of the six frames.
+                contextWindow: 200000
+                imageRequestTokens: 384
+                imageRequestMaxBytes: 300
+
+- id: attachment-local
+  name: '@deepseek-ai/dsh-attachment-local'

+ 33 - 0
snapshots/acp/image-offload/cordis.yml

@@ -0,0 +1,33 @@
+# Image-offload overlay: the image scenario with the shipped vision model and
+# the durable attachment store the read_image tool commits through. The store
+# resolves its root from $DSH_HOME, which the snapshot harness scopes per run,
+# so the patch itself carries no attachment path. No compaction budget is
+# tightened: the route's request-image budget alone drives the `image/offload`
+# watermark this scenario pins.
+- id: acp
+  name: '@deepseek-ai/dsh-acp'
+  config:
+    provider: deepseek-official
+    model: deepseek-v4-flash-vision-exp
+
+- id: session-persistence-jsonl
+  name: '@deepseek-ai/dsh-session-persistence-jsonl'
+  config:
+    root: !!js process.env.DSH_SNAPSHOT_SESSIONS_ROOT ?? './.sessions'
+    compression: none
+
+- id: agent-instructions
+  name: '@deepseek-ai/dsh-agent-instructions'
+  config:
+    maxBytes: 65536
+
+- id: system-prompt
+  name: '@deepseek-ai/dsh-system-prompt'
+  config:
+    persona: |
+      You are a coding assistant powered by the {{model}} model. Your working directory is {{cwd}}. Your bash tool runs under a file sandbox — a `[sandbox: file access denied …]` result is policy, not a command bug.
+
+      Verify your work by running the code or tests. Keep answers brief and factual.
+
+- id: attachment-local
+  name: '@deepseek-ai/dsh-attachment-local'

+ 86 - 0
snapshots/acp/image-offload/input.json

@@ -0,0 +1,86 @@
+{
+  "steps": [
+    {
+      "op": "initialize"
+    },
+    {
+      "op": "newSession"
+    },
+    {
+      "op": "promptContent",
+      "content": [
+        {
+          "type": "text",
+          "text": "Here are six reference screenshots of the dashboard: "
+        },
+        {
+          "type": "image",
+          "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
+          "mimeType": "image/png"
+        },
+        {
+          "type": "text",
+          "text": " (frame 1) "
+        },
+        {
+          "type": "image",
+          "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
+          "mimeType": "image/png"
+        },
+        {
+          "type": "text",
+          "text": " (frame 2) "
+        },
+        {
+          "type": "image",
+          "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
+          "mimeType": "image/png"
+        },
+        {
+          "type": "text",
+          "text": " (frame 3) "
+        },
+        {
+          "type": "image",
+          "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
+          "mimeType": "image/png"
+        },
+        {
+          "type": "text",
+          "text": " (frame 4) "
+        },
+        {
+          "type": "image",
+          "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
+          "mimeType": "image/png"
+        },
+        {
+          "type": "text",
+          "text": " (frame 5) "
+        },
+        {
+          "type": "image",
+          "data": "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
+          "mimeType": "image/png"
+        },
+        {
+          "type": "text",
+          "text": " (frame 6) "
+        },
+        {
+          "type": "text",
+          "text": "Acknowledge receipt briefly; we will discuss them next."
+        }
+      ]
+    },
+    {
+      "op": "promptContent",
+      "content": [
+        {
+          "type": "text",
+          "text": "Now reply with exactly the single word DONE."
+        }
+      ]
+    }
+  ]
+}

+ 33 - 0
snapshots/acp/image-offload/session.jsonl

@@ -0,0 +1,33 @@
+{"type":"session","version":0,"id":"{{session:1}}","createdAt":1783952000000,"cwd":"{{cwd}}","delegationDepth":0}
+{"type":"permission/preset","data":{"preset":"danger-full-access"}}
+{"type":"sandbox/mode","data":{"mode":"danger-full-access"}}
+{"type":"approval/policy","data":{"policy":"never"}}
+{"type":"agent/inbox/spliced","data":{"target":"next-turn","start":0,"inserted":[{"content":[{"type":"text","text":"Here are six reference screenshots of the dashboard: "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 1) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 2) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 3) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 4) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 5) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 6) Acknowledge receipt briefly; we will discuss them next."}],"source":{"kind":"user"},"role":"user","id":"{{message:1}}"}]}}
+{"type":"turn/start","data":{"turn":1}}
+{"type":"agent/inbox/spliced","data":{"target":"next-turn","start":0,"removedCount":1,"inserted":[]}}
+{"type":"step/start","data":{"turn":1,"step":1}}
+{"type":"user/message","data":{"content":[{"type":"text","text":"Here are six reference screenshots of the dashboard: "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 1) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 2) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 3) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 4) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 5) "},{"type":"image","attachment":{"attachmentId":"sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640","mediaType":"image/png","width":1,"height":1,"bytes":69}},{"type":"text","text":" (frame 6) Acknowledge receipt briefly; we will discuss them next."}],"source":{"kind":"user"},"role":"user","id":"{{message:1}}"},"surfaceOp":"append"}
+{"type":"user/message","data":{"content":[{"type":"text","text":"Current runtime context. This snapshot supersedes earlier runtime-context snapshots.\n\nCurrent DSH file policy: danger-full-access. The DSH file sandbox does not restrict file modifications by available operations.\n\nApproval prompts are disabled in this session: actions that require approval are rejected automatically — do not request sandbox escalation (do not set `sandbox_permissions`)."}],"source":{"kind":"plugin","plugin":"@deepseek-ai/dsh-system-prompt","form":"snapshot","sections":[{"name":"sandbox:policy","text":"Current DSH file policy: danger-full-access. The DSH file sandbox does not restrict file modifications by available operations."},{"name":"approval:policy","text":"Approval prompts are disabled in this session: actions that require approval are rejected automatically — do not request sandbox escalation (do not set `sandbox_permissions`)."}]},"role":"user","id":"{{message:2}}"},"surfaceOp":"append"}
+{"type":"session/title","data":{"title":"Here are six reference screenshots","messageSeqs":[7],"source":{"kind":"fallback"}}}
+{"type":"request/header","data":{"header":{"config":{"provider":"deepseek-official","model":"deepseek-v4-flash-vision-exp"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}}
+{"type":"request/context","data":{"provider":"deepseek-official","model":"deepseek-v4-flash-vision-exp","contextWindow":200000}}
+{"type":"image/offload","data":{"turn":1,"step":1,"watermark":{"seq":7,"path":[5]}}}
+{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"block-start","index":0,"blockType":"text"}}}
+{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"text","text":"Received six dashboard frames; three of them are already offloaded."}}}}
+{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":3,"outputTokens":3}}}}
+{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"stop"}}}}
+{"type":"assistant/message","data":{"turn":1,"step":1,"message":{"role":"assistant","content":[{"type":"text","text":"Received six dashboard frames; three of them are already offloaded."}],"source":{"kind":"model","provider":"deepseek-official","model":"deepseek-v4-flash-vision-exp"},"id":"{{message:3}}"},"usage":{"inputTokens":3,"outputTokens":3}},"sourceEventSeqs":[[13,16]],"surfaceOp":"append"}
+{"type":"step/end","data":{"turn":1,"step":1}}
+{"type":"turn/end","data":{"turn":1,"reason":{"kind":"completed"}}}
+{"type":"agent/inbox/spliced","data":{"target":"next-turn","start":0,"inserted":[{"content":[{"type":"text","text":"Now reply with exactly the single word DONE."}],"source":{"kind":"user"},"role":"user","id":"{{message:4}}"}]}}
+{"type":"turn/start","data":{"turn":2}}
+{"type":"agent/inbox/spliced","data":{"target":"next-turn","start":0,"removedCount":1,"inserted":[]}}
+{"type":"step/start","data":{"turn":2,"step":1}}
+{"type":"user/message","data":{"content":[{"type":"text","text":"Now reply with exactly the single word DONE."}],"source":{"kind":"user"},"role":"user","id":"{{message:4}}"},"surfaceOp":"append"}
+{"type":"assistant/chunk","data":{"turn":2,"step":1,"chunk":{"type":"block-start","index":0,"blockType":"text"}}}
+{"type":"assistant/chunk","data":{"turn":2,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"text","text":"DONE"}}}}
+{"type":"assistant/chunk","data":{"turn":2,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":3,"outputTokens":3}}}}
+{"type":"assistant/chunk","data":{"turn":2,"step":1,"chunk":{"type":"finish","reason":{"kind":"stop"}}}}
+{"type":"assistant/message","data":{"turn":2,"step":1,"message":{"role":"assistant","content":[{"type":"text","text":"DONE"}],"source":{"kind":"model","provider":"deepseek-official","model":"deepseek-v4-flash-vision-exp"},"id":"{{message:5}}"},"usage":{"inputTokens":3,"outputTokens":3}},"sourceEventSeqs":[[25,28]],"surfaceOp":"append"}
+{"type":"step/end","data":{"turn":2,"step":1}}
+{"type":"turn/end","data":{"turn":2,"reason":{"kind":"completed"}}}

+ 16 - 0
snapshots/acp/image-offload/snapshot.yml

@@ -0,0 +1,16 @@
+version: 1
+scenario: image-offload
+profile: acp
+composition: image-offload
+recording: authored
+header:
+  class: image-offload
+  pin: true
+  systemPromptSource: session/read-image
+  toolSchemasSource: escalation-approved
+permission: danger-full-access
+input:
+  attachments:
+    - id: sha256:b1ff9c8ea3a780bad09b346c423d2d0e46815926879b18e841d928376a946640
+      mediaType: image/png
+      data: iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElFTkSuQmCC

+ 8 - 0
snapshots/acp/image-offload/stdout.expected.jsonl

@@ -0,0 +1,8 @@
+{"jsonrpc":"2.0","id":1,"result":{"protocolVersion":1,"agentInfo":{"name":"deepseek-harness-acp","version":"0.0.1"},"agentCapabilities":{"mcpCapabilities":{"http":true},"promptCapabilities":{"image":true,"audio":false,"embeddedContext":false},"sessionCapabilities":{"close":{},"list":{},"resume":{}}},"authMethods":[]}}
+{"jsonrpc":"2.0","id":2,"result":{"sessionId":"{{sessionId}}","configOptions":[{"id":"model","name":"Model","category":"model","type":"select","currentValue":"[\"deepseek-official\",\"deepseek-v4-flash-vision-exp\"]","options":[{"group":"deepseek-official","name":"DeepSeek","options":[{"value":"[\"deepseek-official\",\"deepseek-v4-flash\"]","name":"deepseek-v4-flash"},{"value":"[\"deepseek-official\",\"deepseek-v4-pro\"]","name":"deepseek-v4-pro"},{"value":"[\"deepseek-official\",\"deepseek-v4-flash-vision-exp\"]","name":"deepseek-v4-flash-vision-exp"}]}]}]}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","messageId":"{{messageId}}","content":{"type":"text","text":"Received six dashboard frames; three of them are already offloaded."}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"usage_update","used":"{{usedTokens}}","size":200000}}}
+{"jsonrpc":"2.0","id":3,"result":{"stopReason":"end_turn"}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"agent_message_chunk","messageId":"{{messageId}}","content":{"type":"text","text":"DONE"}}}}
+{"jsonrpc":"2.0","method":"session/update","params":{"sessionId":"{{sessionId}}","update":{"sessionUpdate":"usage_update","used":"{{usedTokens}}","size":200000}}}
+{"jsonrpc":"2.0","id":4,"result":{"stopReason":"end_turn"}}

+ 1 - 0
snapshots/acp/image-offload/system-prompt.expected.md

@@ -0,0 +1 @@
+../../session/read-image/system-prompt.expected.md