Răsfoiți Sursa

merge: PR #48 统一上下文预算到 token 域并增加输出上限

Mochocyang 1 lună în urmă
părinte
comite
4580dac8f1
41 a modificat fișierele cu 2110 adăugiri și 475 ștergeri
  1. 2 2
      package-lock.json
  2. 2 2
      src/App.tsx
  3. 2 2
      src/components/chat/chat-model-selector.tsx
  4. 18 15
      src/components/settings/context-size-selector.tsx
  5. 46 24
      src/components/settings/llm-presets.ts
  6. 24 1
      src/components/settings/llm-wiki-model-settings.spec.ts
  7. 87 0
      src/components/settings/output-tokens-selector.tsx
  8. 36 1
      src/components/settings/preset-resolver.spec.ts
  9. 16 6
      src/components/settings/preset-resolver.ts
  10. 38 5
      src/components/settings/sections/custom-provider-cards.tsx
  11. 58 4
      src/components/settings/sections/llm-provider-section.tsx
  12. 56 7
      src/components/sources/outline-chat-panel.tsx
  13. 9 1
      src/i18n/en.json
  14. 9 1
      src/i18n/zh.json
  15. 16 2
      src/lib/agent/runner.ts
  16. 2 2
      src/lib/agent/tools/index.ts
  17. 83 1
      src/lib/chat-request-budget.test.ts
  18. 270 0
      src/lib/chat-request-budget.ts
  19. 16 22
      src/lib/context-budget.contract.spec.ts
  20. 3 3
      src/lib/context-budget.spec.ts
  21. 216 73
      src/lib/context-budget.test.ts
  22. 242 140
      src/lib/context-budget.ts
  23. 6 3
      src/lib/context-hub/ai-outline-integration.spec.ts
  24. 1 1
      src/lib/context-hub/composer.ts
  25. 1 1
      src/lib/context-hub/types.ts
  26. 13 1
      src/lib/env-llm-defaults.ts
  27. 22 21
      src/lib/ingest.prompt.test.ts
  28. 25 18
      src/lib/ingest.ts
  29. 87 15
      src/lib/llm-client.ts
  30. 148 5
      src/lib/llm-client.usage.spec.ts
  31. 72 0
      src/lib/llm-context-size.ts
  32. 85 15
      src/lib/llm-providers.spec.ts
  33. 73 40
      src/lib/llm-providers.ts
  34. 50 9
      src/lib/novel/context-engine.spec.ts
  35. 7 2
      src/lib/novel/context-engine.ts
  36. 44 3
      src/lib/novel/deep-chapter-generation.spec.ts
  37. 42 14
      src/lib/novel/deep-chapter-generation.ts
  38. 11 5
      src/lib/novel/model-resolver.ts
  39. 76 1
      src/lib/project-store.integration.test.ts
  40. 82 4
      src/lib/project-store.ts
  41. 14 3
      src/stores/wiki-store.ts

+ 2 - 2
package-lock.json

@@ -1,12 +1,12 @@
 {
 {
   "name": "qmai",
   "name": "qmai",
-  "version": "3.1.0",
+  "version": "3.1.5",
   "lockfileVersion": 3,
   "lockfileVersion": 3,
   "requires": true,
   "requires": true,
   "packages": {
   "packages": {
     "": {
     "": {
       "name": "qmai",
       "name": "qmai",
-      "version": "3.1.0",
+      "version": "3.1.5",
       "license": "GPL-3.0-or-later",
       "license": "GPL-3.0-or-later",
       "dependencies": {
       "dependencies": {
         "@base-ui/react": "^1.7.0",
         "@base-ui/react": "^1.7.0",

+ 2 - 2
src/App.tsx

@@ -17,7 +17,7 @@ import { WelcomeScreen } from "@/components/project/welcome-screen"
 import { CreateProjectDialog } from "@/components/project/create-project-dialog"
 import { CreateProjectDialog } from "@/components/project/create-project-dialog"
 import { formatAppTitle } from "@/lib/app-title"
 import { formatAppTitle } from "@/lib/app-title"
 import { resetProjectState } from "@/lib/reset-project-state"
 import { resetProjectState } from "@/lib/reset-project-state"
-import { LLM_PRESETS } from "@/components/settings/llm-presets"
+import { findLlmPresetById } from "@/components/settings/llm-presets"
 import { resolveConfig } from "@/components/settings/preset-resolver"
 import { resolveConfig } from "@/components/settings/preset-resolver"
 import { toast } from "@/lib/toast"
 import { toast } from "@/lib/toast"
 import type { WikiProject } from "@/types/wiki"
 import type { WikiProject } from "@/types/wiki"
@@ -303,7 +303,7 @@ function App() {
           // `llmConfig` snapshot from a previous launch would keep the
           // `llmConfig` snapshot from a previous launch would keep the
           // old value. Overrides still win, so an explicit user choice
           // old value. Overrides still win, so an explicit user choice
           // is preserved.
           // is preserved.
-          const preset = LLM_PRESETS.find((p) => p.id === savedActivePreset)
+          const preset = findLlmPresetById(savedActivePreset)
           if (preset) {
           if (preset) {
             const currentFallback = useWikiStore.getState().llmConfig
             const currentFallback = useWikiStore.getState().llmConfig
             const override = (savedProviderConfigs ?? {})[savedActivePreset]
             const override = (savedProviderConfigs ?? {})[savedActivePreset]

+ 2 - 2
src/components/chat/chat-model-selector.tsx

@@ -4,7 +4,7 @@ import { ChevronDown, Check } from "lucide-react"
 import { createPortal } from "react-dom"
 import { createPortal } from "react-dom"
 import { Button } from "@/components/ui/button"
 import { Button } from "@/components/ui/button"
 import { useWikiStore, type SavedModel } from "@/stores/wiki-store"
 import { useWikiStore, type SavedModel } from "@/stores/wiki-store"
-import { LLM_PRESETS } from "@/components/settings/llm-presets"
+import { findLlmPresetById } from "@/components/settings/llm-presets"
 import { getEffectiveSavedModels, isProviderAvailable } from "@/lib/llm-model-keys"
 import { getEffectiveSavedModels, isProviderAvailable } from "@/lib/llm-model-keys"
 
 
 interface ChatModelSelectorProps {
 interface ChatModelSelectorProps {
@@ -85,7 +85,7 @@ export function ChatModelSelector({ value, onChange, disabled }: ChatModelSelect
       if (!isProviderAvailable(key, config)) continue
       if (!isProviderAvailable(key, config)) continue
       const models = getEffectiveSavedModels(config)
       const models = getEffectiveSavedModels(config)
       if (models.length > 0) {
       if (models.length > 0) {
-        const preset = LLM_PRESETS.find((p) => p.id === key)
+        const preset = findLlmPresetById(key)
         groups.push({
         groups.push({
           id: key,
           id: key,
           label: preset?.label || config.label || key,
           label: preset?.label || config.label || key,

+ 18 - 15
src/components/settings/context-size-selector.tsx

@@ -1,20 +1,19 @@
-const CONTEXT_PRESETS = [
-  { value: 4096, label: "4K" },
-  { value: 8192, label: "8K" },
-  { value: 16384, label: "16K" },
-  { value: 32768, label: "32K" },
-  { value: 65536, label: "64K" },
-  { value: 131072, label: "128K" },
+import { useTranslation } from "react-i18next"
+import { normalizeUserLlmContextSize } from "@/lib/llm-context-size"
+
+export const CONTEXT_PRESETS = [
   { value: 204800, label: "200K" },
   { value: 204800, label: "200K" },
   { value: 262144, label: "256K" },
   { value: 262144, label: "256K" },
+  { value: 307200, label: "300K" },
+  { value: 409600, label: "400K" },
   { value: 524288, label: "512K" },
   { value: 524288, label: "512K" },
   { value: 1000000, label: "1M" },
   { value: 1000000, label: "1M" },
 ]
 ]
 
 
-function formatSize(chars: number): string {
-  if (chars >= 1000000) return `${(chars / 1000000).toFixed(1)}M characters`
-  if (chars >= 1000) return `${Math.round(chars / 1000)}K characters`
-  return `${chars} characters`
+function formatSize(tokens: number): string {
+  if (tokens >= 1_000_000) return `${(tokens / 1_000_000).toFixed(1)}M`
+  if (tokens >= 1024) return `${Math.round(tokens / 1024)}K`
+  return String(tokens)
 }
 }
 
 
 export function ContextSizeSelector({
 export function ContextSizeSelector({
@@ -24,8 +23,10 @@ export function ContextSizeSelector({
   value: number
   value: number
   onChange: (v: number) => void
   onChange: (v: number) => void
 }) {
 }) {
+  const { t } = useTranslation()
+  const normalizedValue = normalizeUserLlmContextSize(value)
   const closestIndex = CONTEXT_PRESETS.reduce((best, preset, i) => {
   const closestIndex = CONTEXT_PRESETS.reduce((best, preset, i) => {
-    return Math.abs(preset.value - value) < Math.abs(CONTEXT_PRESETS[best].value - value)
+    return Math.abs(preset.value - normalizedValue) < Math.abs(CONTEXT_PRESETS[best].value - normalizedValue)
       ? i
       ? i
       : best
       : best
   }, 0)
   }, 0)
@@ -34,9 +35,8 @@ export function ContextSizeSelector({
   return (
   return (
     <div>
     <div>
       <div className="flex items-center justify-between mb-2">
       <div className="flex items-center justify-between mb-2">
-        <span className="text-sm font-medium">{formatSize(value)}</span>
-        <span className="text-xs text-muted-foreground">
-          ~{Math.floor((value * 0.6) / 1000)}K chars for wiki content
+        <span className="text-sm font-medium">
+          {t("settings.sections.llm.contextWindowValue", { value: formatSize(normalizedValue) })}
         </span>
         </span>
       </div>
       </div>
       <input
       <input
@@ -65,6 +65,9 @@ export function ContextSizeSelector({
           </button>
           </button>
         ))}
         ))}
       </div>
       </div>
+      <p className="text-[10px] text-muted-foreground mt-1">
+        {t("settings.sections.llm.contextWindowHint")}
+      </p>
     </div>
     </div>
   )
   )
 }
 }

+ 46 - 24
src/components/settings/llm-presets.ts

@@ -1,4 +1,5 @@
 import type { AzureModelFamily } from "@/stores/wiki-store"
 import type { AzureModelFamily } from "@/stores/wiki-store"
+import { MIN_USER_LLM_CONTEXT_SIZE } from "@/lib/llm-context-size"
 
 
 /**
 /**
  * Curated LLM provider presets.
  * Curated LLM provider presets.
@@ -56,8 +57,15 @@ export interface LlmPreset {
   suggestedModels?: string[]
   suggestedModels?: string[]
   /** Custom providers only: which wire protocol to speak. */
   /** Custom providers only: which wire protocol to speak. */
   apiMode?: CustomApiMode
   apiMode?: CustomApiMode
-  /** Suggested context window; user can override. */
+  /** Suggested context window in tokens, from the model's spec sheet; user can override. */
   suggestedContextSize?: number
   suggestedContextSize?: number
+  /**
+   * Suggested maximum output in tokens, from the model's spec sheet; user can
+   * override. Only fill this in where the figure has a source — a wrong value
+   * here either wastes the model's capacity or gets the request rejected.
+   * Omitted presets fall back to `DEFAULT_USER_LLM_MAX_OUTPUT_TOKENS`.
+   */
+  suggestedMaxOutputTokens?: number
 }
 }
 
 
 const RAW_LLM_PRESETS: LlmPreset[] = [
 const RAW_LLM_PRESETS: LlmPreset[] = [
@@ -87,7 +95,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "claude-3-5-sonnet-20241022",
       "claude-3-5-sonnet-20241022",
       "claude-3-5-haiku-20241022",
       "claude-3-5-haiku-20241022",
     ],
     ],
-    suggestedContextSize: 200000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "claude-code-cli",
     id: "claude-code-cli",
@@ -105,7 +113,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "claude-sonnet-4-5-20250929",
       "claude-sonnet-4-5-20250929",
       "claude-haiku-4-5-20251001",
       "claude-haiku-4-5-20251001",
     ],
     ],
-    suggestedContextSize: 200000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "codex-cli",
     id: "codex-cli",
@@ -120,7 +128,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "gpt-5.3-codex-spark",
       "gpt-5.3-codex-spark",
       "gpt-5.2",
       "gpt-5.2",
     ],
     ],
-    suggestedContextSize: 200000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "cursor-cli",
     id: "cursor-cli",
@@ -141,7 +149,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "gpt-5.5-medium",
       "gpt-5.5-medium",
       "claude-opus-4-7-thinking-max",
       "claude-opus-4-7-thinking-max",
     ],
     ],
-    suggestedContextSize: 200000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "openai",
     id: "openai",
@@ -163,7 +171,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "o1-mini",
       "o1-mini",
       "gpt-4-turbo",
       "gpt-4-turbo",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "google",
     id: "google",
@@ -191,7 +199,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     baseUrl: "https://your-resource.openai.azure.com",
     baseUrl: "https://your-resource.openai.azure.com",
     defaultModel: "your-deployment-name",
     defaultModel: "your-deployment-name",
     azureApiVersion: "2024-10-21",
     azureApiVersion: "2024-10-21",
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "deepseek",
     id: "deepseek",
@@ -212,6 +220,8 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "deepseek-reasoner",
       "deepseek-reasoner",
     ],
     ],
     suggestedContextSize: 1000000,
     suggestedContextSize: 1000000,
+    // DeepSeek-V4: 1000K context / 384K max output, per the published spec.
+    suggestedMaxOutputTokens: 393216,
   },
   },
   {
   {
     id: "atlascloud",
     id: "atlascloud",
@@ -240,7 +250,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "openai/gpt-5.5",
       "openai/gpt-5.5",
       "google/gemini-3.5-flash",
       "google/gemini-3.5-flash",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "groq",
     id: "groq",
@@ -250,19 +260,17 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     baseUrl: "https://api.groq.com/openai/v1",
     baseUrl: "https://api.groq.com/openai/v1",
     defaultModel: "llama-3.3-70b-versatile",
     defaultModel: "llama-3.3-70b-versatile",
     apiMode: "chat_completions",
     apiMode: "chat_completions",
-    // Groq hosts open-weight models; list stays current-practical picks.
+    // Writing workflows require at least 204800 tokens of context.
     suggestedModels: [
     suggestedModels: [
       "llama-3.3-70b-versatile",
       "llama-3.3-70b-versatile",
       "llama-3.1-8b-instant",
       "llama-3.1-8b-instant",
       "llama-3.1-70b-versatile",
       "llama-3.1-70b-versatile",
-      "mixtral-8x7b-32768",
-      "gemma2-9b-it",
       "moonshotai/kimi-k2-instruct",
       "moonshotai/kimi-k2-instruct",
       "openai/gpt-oss-120b",
       "openai/gpt-oss-120b",
       "openai/gpt-oss-20b",
       "openai/gpt-oss-20b",
       "qwen/qwen3-32b",
       "qwen/qwen3-32b",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "xai",
     id: "xai",
@@ -282,7 +290,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "grok-code-fast-1",
       "grok-code-fast-1",
       "grok-2-vision-1212",
       "grok-2-vision-1212",
     ],
     ],
-    suggestedContextSize: 131072,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "nvidia-nim",
     id: "nvidia-nim",
@@ -319,7 +327,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "mistralai/mixtral-8x22b-instruct",
       "mistralai/mixtral-8x22b-instruct",
       "mistralai/mistral-large-2-instruct",
       "mistralai/mistral-large-2-instruct",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "kimi",
     id: "kimi",
@@ -407,7 +415,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "glm-4v-plus",
       "glm-4v-plus",
       "glm-zero-preview",
       "glm-zero-preview",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "minimax-global",
     id: "minimax-global",
@@ -422,7 +430,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     // dropped — users who still need them can type the id into the
     // dropped — users who still need them can type the id into the
     // custom input.
     // custom input.
     suggestedModels: ["MiniMax-M3", "MiniMax-M2.7"],
     suggestedModels: ["MiniMax-M3", "MiniMax-M2.7"],
-    suggestedContextSize: 200000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "minimax-cn",
     id: "minimax-cn",
@@ -433,7 +441,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     defaultModel: "MiniMax-M3",
     defaultModel: "MiniMax-M3",
     apiMode: "anthropic_messages",
     apiMode: "anthropic_messages",
     suggestedModels: ["MiniMax-M3", "MiniMax-M2.7"],
     suggestedModels: ["MiniMax-M3", "MiniMax-M2.7"],
-    suggestedContextSize: 200000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "bailian-coding",
     id: "bailian-coding",
@@ -466,7 +474,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "qwen3-coder-next",
       "qwen3-coder-next",
       "glm-4.7",
       "glm-4.7",
     ],
     ],
-    suggestedContextSize: 131072,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "xiaomi-mimo",
     id: "xiaomi-mimo",
@@ -521,7 +529,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "GLM-4.7",
       "GLM-4.7",
       "DeepSeek-V3",
       "DeepSeek-V3",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "ollama-local",
     id: "ollama-local",
@@ -531,7 +539,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     baseUrl: "http://localhost:11434",
     baseUrl: "http://localhost:11434",
     // Intentionally no suggestedModels: local set depends on what the
     // Intentionally no suggestedModels: local set depends on what the
     // user has actually pulled / loaded. Kept as free-text input.
     // user has actually pulled / loaded. Kept as free-text input.
-    suggestedContextSize: 32768,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "ollama-cloud",
     id: "ollama-cloud",
@@ -548,7 +556,7 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
       "kimi-k2:1t",
       "kimi-k2:1t",
       "deepseek-v3.1:671b",
       "deepseek-v3.1:671b",
     ],
     ],
-    suggestedContextSize: 128000,
+    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
   },
   {
   {
     id: "custom",
     id: "custom",
@@ -563,9 +571,23 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
   },
   },
 ]
 ]
 
 
-export const LLM_PRESETS: LlmPreset[] = RAW_LLM_PRESETS.filter(
-  (preset, index) => preset.id !== "custom" || index === 0,
-)
+const ALL_LLM_PRESETS: LlmPreset[] = RAW_LLM_PRESETS
+  .filter((preset, index) => preset.id !== "custom" || index === 0)
+  .map((preset) => ({
+    ...preset,
+    suggestedContextSize: Math.max(
+      preset.suggestedContextSize ?? MIN_USER_LLM_CONTEXT_SIZE,
+      MIN_USER_LLM_CONTEXT_SIZE,
+    ),
+  }))
+
+/** All providers remain visible and default to at least the writing floor. */
+export const LLM_PRESETS: LlmPreset[] = ALL_LLM_PRESETS
+
+/** Resolve provider configurations by their stable preset id. */
+export function findLlmPresetById(id: string): LlmPreset | undefined {
+  return ALL_LLM_PRESETS.find((preset) => preset.id === id)
+}
 
 
 /**
 /**
  * Best-effort reverse lookup: given the current LlmConfig fields, which
  * Best-effort reverse lookup: given the current LlmConfig fields, which

+ 24 - 1
src/components/settings/llm-wiki-model-settings.spec.ts

@@ -1,4 +1,6 @@
 import { describe, expect, it } from "vitest"
 import { describe, expect, it } from "vitest"
+import { readFileSync } from "node:fs"
+import { resolve } from "node:path"
 import { getProviderConfig } from "@/lib/llm-providers"
 import { getProviderConfig } from "@/lib/llm-providers"
 import type { LlmConfig } from "@/stores/wiki-store"
 import type { LlmConfig } from "@/stores/wiki-store"
 import zh from "@/i18n/zh.json"
 import zh from "@/i18n/zh.json"
@@ -23,7 +25,16 @@ function preset(id: string) {
 }
 }
 
 
 describe("QMAI model settings", () => {
 describe("QMAI model settings", () => {
-  it("includes the built-in provider rows below the custom row", () => {
+  it("renders the 200K long-writing requirement on the LLM settings page", () => {
+    const source = readFileSync(
+      resolve(__dirname, "sections/llm-provider-section.tsx"),
+      "utf8",
+    )
+    expect(source).toContain("settings.sections.llm.longWritingContextTitle")
+    expect(source).toContain("settings.sections.llm.longWritingContextHint")
+  })
+
+  it("keeps every built-in provider and gives each one at least 204800", () => {
     expect(LLM_PRESETS[0]?.id).toBe("custom")
     expect(LLM_PRESETS[0]?.id).toBe("custom")
 
 
     expect(LLM_PRESETS.map((item) => item.id)).toEqual([
     expect(LLM_PRESETS.map((item) => item.id)).toEqual([
@@ -52,6 +63,16 @@ describe("QMAI model settings", () => {
       "ollama-local",
       "ollama-local",
       "ollama-cloud",
       "ollama-cloud",
     ])
     ])
+    for (const item of LLM_PRESETS) {
+      expect(item.suggestedContextSize).toBeGreaterThanOrEqual(204_800)
+    }
+  })
+
+  it("removes only known models below the writing window from suggestions", () => {
+    const groq = preset("groq")
+    expect(groq.suggestedModels).not.toContain("mixtral-8x7b-32768")
+    expect(groq.suggestedModels).not.toContain("gemma2-9b-it")
+    expect(groq.defaultModel).toBe("llama-3.3-70b-versatile")
   })
   })
 
 
   it("resolves DeepSeek to an OpenAI-compatible custom endpoint", () => {
   it("resolves DeepSeek to an OpenAI-compatible custom endpoint", () => {
@@ -128,6 +149,8 @@ describe("QMAI model settings", () => {
     expect(llm.collapse).toBe("收起配置")
     expect(llm.collapse).toBe("收起配置")
     expect(llm.apiKeyPlaceholder).toBe("输入 API Key")
     expect(llm.apiKeyPlaceholder).toBe("输入 API Key")
     expect(llm.activeBadge).toBe("当前使用")
     expect(llm.activeBadge).toBe("当前使用")
+    expect(llm.longWritingContextTitle).toContain("至少 200K")
+    expect(llm.longWritingContextHint).toContain("204800")
     expect(JSON.stringify(llm)).not.toContain("??")
     expect(JSON.stringify(llm)).not.toContain("??")
   })
   })
 })
 })

+ 87 - 0
src/components/settings/output-tokens-selector.tsx

@@ -0,0 +1,87 @@
+import { useTranslation } from "react-i18next"
+import { normalizeUserLlmMaxOutputTokens } from "@/lib/llm-context-size"
+
+export const OUTPUT_TOKEN_PRESETS = [
+  { value: 65536, label: "64K" },
+  { value: 131072, label: "128K" },
+  { value: 262144, label: "256K" },
+  { value: 393216, label: "384K" },
+]
+
+function formatTokens(tokens: number): string {
+  if (tokens >= 1000) return `${Math.round(tokens / 1024)}K`
+  return String(tokens)
+}
+
+/**
+ * Declares how much the selected model can emit in one response — a capability
+ * ceiling, not a request size. Each workflow asks for what it needs and this
+ * only ever caps it, so raising the slider does not make responses longer.
+ */
+export function OutputTokensSelector({
+  value,
+  contextWindow,
+  onChange,
+}: {
+  value: number | undefined
+  contextWindow?: number
+  onChange: (v: number) => void
+}) {
+  const { t } = useTranslation()
+  const normalizedValue = normalizeUserLlmMaxOutputTokens(value)
+  const closestIndex = OUTPUT_TOKEN_PRESETS.reduce((best, preset, i) => {
+    return Math.abs(preset.value - normalizedValue) < Math.abs(OUTPUT_TOKEN_PRESETS[best].value - normalizedValue)
+      ? i
+      : best
+  }, 0)
+  const pct = (closestIndex / (OUTPUT_TOKEN_PRESETS.length - 1)) * 100
+  // Output and input share one window. Spec sheets often list both as the same
+  // size (e.g. Doubao 256K/256K); that is a valid capability, not overflow.
+  // Only warn when the declared ceiling is strictly larger than the window.
+  const exceedsWindow =
+    typeof contextWindow === "number" && contextWindow > 0 && normalizedValue > contextWindow
+
+  return (
+    <div>
+      <div className="flex items-center justify-between mb-2">
+        <span className="text-sm font-medium">
+          {t("settings.sections.llm.maxOutputTokensValue", { value: formatTokens(normalizedValue) })}
+        </span>
+        {exceedsWindow ? (
+          <span className="text-xs text-amber-600 dark:text-amber-500">
+            {t("settings.sections.llm.maxOutputTokensExceedsWindow")}
+          </span>
+        ) : null}
+      </div>
+      <input
+        type="range"
+        min={0}
+        max={OUTPUT_TOKEN_PRESETS.length - 1}
+        step={1}
+        value={closestIndex}
+        onChange={(e) => onChange(OUTPUT_TOKEN_PRESETS[parseInt(e.target.value)].value)}
+        className="w-full h-2 rounded-lg appearance-none cursor-pointer accent-primary"
+        style={{
+          background: `linear-gradient(to right, #4f46e5 ${pct}%, #e5e7eb ${pct}%)`,
+        }}
+      />
+      <div className="flex justify-between mt-1">
+        {OUTPUT_TOKEN_PRESETS.map((preset, i) => (
+          <button
+            key={preset.value}
+            type="button"
+            onClick={() => onChange(preset.value)}
+            className={`text-[9px] px-0.5 ${
+              i === closestIndex ? "text-primary font-bold" : "text-muted-foreground/50"
+            }`}
+          >
+            {preset.label}
+          </button>
+        ))}
+      </div>
+      <p className="text-[10px] text-muted-foreground mt-1">
+        {t("settings.sections.llm.maxOutputTokensHint")}
+      </p>
+    </div>
+  )
+}

+ 36 - 1
src/components/settings/preset-resolver.spec.ts

@@ -1,6 +1,6 @@
 import { describe, expect, it } from "vitest"
 import { describe, expect, it } from "vitest"
 import { resolveConfig } from "./preset-resolver"
 import { resolveConfig } from "./preset-resolver"
-import type { LlmPreset } from "./llm-presets"
+import { LLM_PRESETS, type LlmPreset } from "./llm-presets"
 import type { LlmConfig } from "@/stores/wiki-store"
 import type { LlmConfig } from "@/stores/wiki-store"
 
 
 const fallback: LlmConfig = {
 const fallback: LlmConfig = {
@@ -36,3 +36,38 @@ describe("resolveConfig functionCallingEnabled", () => {
     expect(config.functionCallingEnabled).toBe(false)
     expect(config.functionCallingEnabled).toBe(false)
   })
   })
 })
 })
+
+describe("resolveConfig context and output limits", () => {
+  const deepseekPreset = LLM_PRESETS.find((preset) => preset.id === "deepseek")!
+
+  it("keeps the DeepSeek suggestion when the user has not chosen a window", () => {
+    const config = resolveConfig(deepseekPreset, { apiKey: "sk" }, fallback)
+    expect(config.maxContextSize).toBe(1_000_000)
+    expect(config.maxOutputTokens).toBe(393_216)
+  })
+
+  it("does not override a window the user set explicitly", () => {
+    // The window used to be forced back up to 1M here, so the DeepSeek slider
+    // looked adjustable but never took effect.
+    const config = resolveConfig(
+      deepseekPreset,
+      { apiKey: "sk", maxContextSize: 262_144 },
+      fallback,
+    )
+    expect(config.maxContextSize).toBe(262_144)
+  })
+
+  it("falls back to the default output limit for presets without a published figure", () => {
+    const config = resolveConfig(customPreset, { apiKey: "sk", model: "m" }, fallback)
+    expect(config.maxOutputTokens).toBe(131_072)
+  })
+
+  it("prefers an explicit output limit over the preset suggestion", () => {
+    const config = resolveConfig(
+      deepseekPreset,
+      { apiKey: "sk", maxOutputTokens: 65_536 },
+      fallback,
+    )
+    expect(config.maxOutputTokens).toBe(65_536)
+  })
+})

+ 16 - 6
src/components/settings/preset-resolver.ts

@@ -1,8 +1,11 @@
 import type { LlmConfig } from "@/stores/wiki-store"
 import type { LlmConfig } from "@/stores/wiki-store"
 import type { ProviderOverride } from "@/stores/wiki-store"
 import type { ProviderOverride } from "@/stores/wiki-store"
 import { AZURE_OPENAI_API_VERSION } from "@/lib/azure-openai"
 import { AZURE_OPENAI_API_VERSION } from "@/lib/azure-openai"
-import { getEffectiveMaxContextSize } from "@/lib/llm-providers"
 import type { LlmPreset } from "./llm-presets"
 import type { LlmPreset } from "./llm-presets"
+import {
+  normalizeUserLlmContextSize,
+  normalizeUserLlmMaxOutputTokens,
+} from "@/lib/llm-context-size"
 
 
 /**
 /**
  * Build a full LlmConfig from a preset template + the user's saved
  * Build a full LlmConfig from a preset template + the user's saved
@@ -17,8 +20,12 @@ export function resolveConfig(
   const ov = override ?? {}
   const ov = override ?? {}
   const apiKey = ov.apiKey ?? ""
   const apiKey = ov.apiKey ?? ""
   const model = ov.model?.trim() || preset.defaultModel || ""
   const model = ov.model?.trim() || preset.defaultModel || ""
-  const rawMaxContextSize =
-    ov.maxContextSize ?? preset.suggestedContextSize ?? fallback.maxContextSize
+  const rawMaxContextSize = normalizeUserLlmContextSize(
+    ov.maxContextSize ?? preset.suggestedContextSize ?? fallback.maxContextSize,
+  )
+  const rawMaxOutputTokens = normalizeUserLlmMaxOutputTokens(
+    ov.maxOutputTokens ?? preset.suggestedMaxOutputTokens ?? fallback.maxOutputTokens,
+  )
   const reasoning = ov.reasoning ?? { mode: "auto" as const }
   const reasoning = ov.reasoning ?? { mode: "auto" as const }
   const localCliIsolation = ov.localCliIsolation === true
   const localCliIsolation = ov.localCliIsolation === true
   const functionCallingEnabled = ov.functionCallingEnabled !== false
   const functionCallingEnabled = ov.functionCallingEnabled !== false
@@ -37,6 +44,7 @@ export function resolveConfig(
       ollamaUrl: fallback.ollamaUrl,
       ollamaUrl: fallback.ollamaUrl,
       customEndpoint: ov.baseUrl ?? preset.baseUrl ?? "",
       customEndpoint: ov.baseUrl ?? preset.baseUrl ?? "",
       maxContextSize: rawMaxContextSize,
       maxContextSize: rawMaxContextSize,
+      maxOutputTokens: rawMaxOutputTokens,
       apiMode: ov.apiMode ?? preset.apiMode ?? "chat_completions",
       apiMode: ov.apiMode ?? preset.apiMode ?? "chat_completions",
       reasoning,
       reasoning,
       localCliIsolation: false,
       localCliIsolation: false,
@@ -50,6 +58,7 @@ export function resolveConfig(
       ollamaUrl: ov.baseUrl ?? preset.baseUrl ?? "http://localhost:11434",
       ollamaUrl: ov.baseUrl ?? preset.baseUrl ?? "http://localhost:11434",
       customEndpoint: fallback.customEndpoint,
       customEndpoint: fallback.customEndpoint,
       maxContextSize: rawMaxContextSize,
       maxContextSize: rawMaxContextSize,
+      maxOutputTokens: rawMaxOutputTokens,
       reasoning,
       reasoning,
       localCliIsolation: false,
       localCliIsolation: false,
       functionCallingEnabled,
       functionCallingEnabled,
@@ -64,6 +73,7 @@ export function resolveConfig(
       azureApiVersion: ov.azureApiVersion ?? preset.azureApiVersion ?? AZURE_OPENAI_API_VERSION,
       azureApiVersion: ov.azureApiVersion ?? preset.azureApiVersion ?? AZURE_OPENAI_API_VERSION,
       azureModelFamily: ov.azureModelFamily ?? preset.azureModelFamily ?? "auto",
       azureModelFamily: ov.azureModelFamily ?? preset.azureModelFamily ?? "auto",
       maxContextSize: rawMaxContextSize,
       maxContextSize: rawMaxContextSize,
+      maxOutputTokens: rawMaxOutputTokens,
       reasoning,
       reasoning,
       localCliIsolation: false,
       localCliIsolation: false,
       functionCallingEnabled,
       functionCallingEnabled,
@@ -80,6 +90,7 @@ export function resolveConfig(
       ollamaUrl: fallback.ollamaUrl,
       ollamaUrl: fallback.ollamaUrl,
       customEndpoint: fallback.customEndpoint,
       customEndpoint: fallback.customEndpoint,
       maxContextSize: rawMaxContextSize,
       maxContextSize: rawMaxContextSize,
+      maxOutputTokens: rawMaxOutputTokens,
       reasoning,
       reasoning,
       localCliIsolation,
       localCliIsolation,
       codexCliTimeoutMinutes: preset.provider === "codex-cli" ? codexCliTimeoutMinutes : undefined,
       codexCliTimeoutMinutes: preset.provider === "codex-cli" ? codexCliTimeoutMinutes : undefined,
@@ -95,6 +106,7 @@ export function resolveConfig(
       ollamaUrl: fallback.ollamaUrl,
       ollamaUrl: fallback.ollamaUrl,
       customEndpoint: ov.baseUrl ?? preset.baseUrl ?? "http://127.0.0.1:8765/v1",
       customEndpoint: ov.baseUrl ?? preset.baseUrl ?? "http://127.0.0.1:8765/v1",
       maxContextSize: rawMaxContextSize,
       maxContextSize: rawMaxContextSize,
+      maxOutputTokens: rawMaxOutputTokens,
       apiMode: "chat_completions",
       apiMode: "chat_completions",
       reasoning,
       reasoning,
       localCliIsolation: false,
       localCliIsolation: false,
@@ -111,14 +123,12 @@ export function resolveConfig(
       ollamaUrl: fallback.ollamaUrl,
       ollamaUrl: fallback.ollamaUrl,
       customEndpoint: fallback.customEndpoint,
       customEndpoint: fallback.customEndpoint,
       maxContextSize: rawMaxContextSize,
       maxContextSize: rawMaxContextSize,
+      maxOutputTokens: rawMaxOutputTokens,
       reasoning,
       reasoning,
       localCliIsolation: false,
       localCliIsolation: false,
       functionCallingEnabled,
       functionCallingEnabled,
     }
     }
   }
   }
 
 
-  // Apply model-specific context size minimums (e.g. DeepSeek → 1M)
-  config.maxContextSize = getEffectiveMaxContextSize(config)
-
   return config
   return config
 }
 }

+ 38 - 5
src/components/settings/sections/custom-provider-cards.tsx

@@ -5,11 +5,21 @@ import { Input } from "@/components/ui/input"
 import { Label } from "@/components/ui/label"
 import { Label } from "@/components/ui/label"
 import { useWikiStore, type ProviderOverride, type SavedModel, type ReasoningConfig } from "@/stores/wiki-store"
 import { useWikiStore, type ProviderOverride, type SavedModel, type ReasoningConfig } from "@/stores/wiki-store"
 import { ContextSizeSelector } from "../context-size-selector"
 import { ContextSizeSelector } from "../context-size-selector"
+import { OutputTokensSelector } from "../output-tokens-selector"
 import { resolveConfig } from "../preset-resolver"
 import { resolveConfig } from "../preset-resolver"
 import { fetchLlmModelList } from "@/lib/settings-model-list"
 import { fetchLlmModelList } from "@/lib/settings-model-list"
 import { useBatchModelTest } from "../hooks/use-batch-model-test"
 import { useBatchModelTest } from "../hooks/use-batch-model-test"
 import { useTranslation } from "react-i18next"
 import { useTranslation } from "react-i18next"
-import { FunctionCallingControls, ReasoningControls } from "./llm-provider-section"
+import {
+  FunctionCallingControls,
+  ReasoningControls,
+  withOutputRoomForReasoning,
+} from "./llm-provider-section"
+import {
+  MIN_USER_LLM_CONTEXT_SIZE,
+  normalizeUserLlmContextSize,
+  normalizeUserLlmMaxOutputTokens,
+} from "@/lib/llm-context-size"
 
 
 interface CustomProviderCard {
 interface CustomProviderCard {
   id: string
   id: string
@@ -19,6 +29,7 @@ interface CustomProviderCard {
   apiKey: string
   apiKey: string
   model: string
   model: string
   maxContextSize?: number
   maxContextSize?: number
+  maxOutputTokens?: number
   reasoning?: ReasoningConfig
   reasoning?: ReasoningConfig
   functionCallingEnabled?: boolean
   functionCallingEnabled?: boolean
   enabled: boolean
   enabled: boolean
@@ -43,7 +54,8 @@ export function CustomProviderCards() {
         baseUrl: config.baseUrl || "",
         baseUrl: config.baseUrl || "",
         apiKey: config.apiKey || "",
         apiKey: config.apiKey || "",
         model: config.model || "",
         model: config.model || "",
-        maxContextSize: config.maxContextSize,
+        maxContextSize: normalizeUserLlmContextSize(config.maxContextSize),
+        maxOutputTokens: config.maxOutputTokens,
         reasoning: config.reasoning,
         reasoning: config.reasoning,
         functionCallingEnabled: config.functionCallingEnabled,
         functionCallingEnabled: config.functionCallingEnabled,
         enabled: config.enabled ?? true,
         enabled: config.enabled ?? true,
@@ -61,6 +73,7 @@ export function CustomProviderCards() {
       baseUrl: "",
       baseUrl: "",
       apiKey: "",
       apiKey: "",
       model: "",
       model: "",
+      maxContextSize: normalizeUserLlmContextSize(undefined),
       enabled: true,
       enabled: true,
       savedModels: [],
       savedModels: [],
     }
     }
@@ -75,6 +88,7 @@ export function CustomProviderCards() {
         baseUrl: newCard.baseUrl,
         baseUrl: newCard.baseUrl,
         apiKey: newCard.apiKey,
         apiKey: newCard.apiKey,
         model: newCard.model,
         model: newCard.model,
+        maxContextSize: newCard.maxContextSize,
         enabled: true,
         enabled: true,
         savedModels: newCard.savedModels,
         savedModels: newCard.savedModels,
       },
       },
@@ -95,7 +109,16 @@ export function CustomProviderCards() {
       baseUrl: updates.baseUrl ?? prev.baseUrl,
       baseUrl: updates.baseUrl ?? prev.baseUrl,
       apiKey: updates.apiKey ?? prev.apiKey,
       apiKey: updates.apiKey ?? prev.apiKey,
       model: updates.model ?? prev.model,
       model: updates.model ?? prev.model,
-      maxContextSize: updates.maxContextSize ?? prev.maxContextSize,
+      maxContextSize: normalizeUserLlmContextSize(
+        updates.maxContextSize ?? prev.maxContextSize,
+      ),
+      ...(updates.maxOutputTokens !== undefined || prev.maxOutputTokens !== undefined
+        ? {
+            maxOutputTokens: normalizeUserLlmMaxOutputTokens(
+              updates.maxOutputTokens ?? prev.maxOutputTokens,
+            ),
+          }
+        : {}),
       reasoning: updates.reasoning ?? prev.reasoning,
       reasoning: updates.reasoning ?? prev.reasoning,
       functionCallingEnabled: updates.functionCallingEnabled ?? prev.functionCallingEnabled,
       functionCallingEnabled: updates.functionCallingEnabled ?? prev.functionCallingEnabled,
       enabled: updates.enabled ?? prev.enabled ?? true,
       enabled: updates.enabled ?? prev.enabled ?? true,
@@ -701,15 +724,25 @@ function CustomProviderCardItem({
           <div className="space-y-2">
           <div className="space-y-2">
             <Label className="text-xs">{t("settings.sections.llm.contextWindow")}</Label>
             <Label className="text-xs">{t("settings.sections.llm.contextWindow")}</Label>
             <ContextSizeSelector
             <ContextSizeSelector
-              value={card.maxContextSize ?? 131072}
+              value={card.maxContextSize ?? MIN_USER_LLM_CONTEXT_SIZE}
               onChange={(v) => onUpdate({ maxContextSize: v })}
               onChange={(v) => onUpdate({ maxContextSize: v })}
             />
             />
           </div>
           </div>
 
 
+          {/* Output ceiling */}
+          <div className="space-y-2">
+            <Label className="text-xs">{t("settings.sections.llm.maxOutputTokens")}</Label>
+            <OutputTokensSelector
+              value={card.maxOutputTokens}
+              contextWindow={card.maxContextSize ?? MIN_USER_LLM_CONTEXT_SIZE}
+              onChange={(v) => onUpdate({ maxOutputTokens: v })}
+            />
+          </div>
+
           {/* Reasoning / thinking */}
           {/* Reasoning / thinking */}
           <ReasoningControls
           <ReasoningControls
             value={card.reasoning ?? { mode: "auto" }}
             value={card.reasoning ?? { mode: "auto" }}
-            onChange={(reasoning) => onUpdate({ reasoning })}
+            onChange={(next) => onUpdate(withOutputRoomForReasoning(next, card.maxOutputTokens))}
           />
           />
 
 
           <FunctionCallingControls
           <FunctionCallingControls

+ 58 - 4
src/components/settings/sections/llm-provider-section.tsx

@@ -7,6 +7,7 @@ import { Label } from "@/components/ui/label"
 import { useWikiStore, type ProviderOverride, type ReasoningConfig, type ReasoningMode, type SavedModel } from "@/stores/wiki-store"
 import { useWikiStore, type ProviderOverride, type ReasoningConfig, type ReasoningMode, type SavedModel } from "@/stores/wiki-store"
 import { LLM_PRESETS, type LlmPreset } from "../llm-presets"
 import { LLM_PRESETS, type LlmPreset } from "../llm-presets"
 import { ContextSizeSelector } from "../context-size-selector"
 import { ContextSizeSelector } from "../context-size-selector"
+import { OutputTokensSelector } from "../output-tokens-selector"
 import { resolveConfig } from "../preset-resolver"
 import { resolveConfig } from "../preset-resolver"
 import { normalizeEndpoint } from "@/lib/endpoint-normalizer"
 import { normalizeEndpoint } from "@/lib/endpoint-normalizer"
 import { isTauri } from "@/lib/platform"
 import { isTauri } from "@/lib/platform"
@@ -17,6 +18,32 @@ import { useBatchModelTest } from "../hooks/use-batch-model-test"
 import { ModelSelectInput } from "../model-select-input"
 import { ModelSelectInput } from "../model-select-input"
 import { SavedModelsManager } from "./saved-models-manager"
 import { SavedModelsManager } from "./saved-models-manager"
 import { CustomProviderCards } from "./custom-provider-cards"
 import { CustomProviderCards } from "./custom-provider-cards"
+import {
+  MIN_USER_LLM_CONTEXT_SIZE,
+  normalizeProviderOverride,
+  normalizeUserLlmMaxOutputTokens,
+} from "@/lib/llm-context-size"
+import { thinkingMinMaxTokens } from "@/lib/llm-providers"
+
+/**
+ * Raise the declared output ceiling when the chosen reasoning level needs more
+ * room than it currently allows.
+ *
+ * Thinking and the final answer share one output allowance. When the ceiling is
+ * too low the request layer drops thinking rather than silently inflating
+ * `max_tokens` past what the model accepts, so the fix belongs here: adjust the
+ * user's own setting, at the moment they change the level, where they can see
+ * and undo it.
+ */
+export function withOutputRoomForReasoning(
+  reasoning: ReasoningConfig,
+  currentMaxOutputTokens: number | undefined,
+): ProviderOverride {
+  const required = thinkingMinMaxTokens(reasoning)
+  const current = normalizeUserLlmMaxOutputTokens(currentMaxOutputTokens)
+  if (required <= current) return { reasoning }
+  return { reasoning, maxOutputTokens: normalizeUserLlmMaxOutputTokens(required) }
+}
 
 
 export function LlmProviderSection() {
 export function LlmProviderSection() {
   const { t } = useTranslation()
   const { t } = useTranslation()
@@ -51,7 +78,10 @@ export function LlmProviderSection() {
   }
   }
 
 
   function updateOverride(id: string, patch: ProviderOverride) {
   function updateOverride(id: string, patch: ProviderOverride) {
-    const merged: ProviderOverride = { ...(providerConfigs[id] ?? {}), ...patch }
+    const merged: ProviderOverride = normalizeProviderOverride({
+      ...(providerConfigs[id] ?? {}),
+      ...patch,
+    })
     const next = { ...providerConfigs, [id]: merged }
     const next = { ...providerConfigs, [id]: merged }
     setProviderConfigs(next)
     setProviderConfigs(next)
     persist(next, activePresetId).catch(() => {})
     persist(next, activePresetId).catch(() => {})
@@ -88,10 +118,22 @@ export function LlmProviderSection() {
         </p>
         </p>
       </div>
       </div>
 
 
+      <div className="flex gap-2 rounded-md border border-amber-500/40 bg-amber-500/10 px-3 py-2 text-sm text-amber-800 dark:text-amber-300">
+        <AlertCircle className="mt-0.5 h-4 w-4 shrink-0" />
+        <div>
+          <div className="font-medium">
+            {t("settings.sections.llm.longWritingContextTitle")}
+          </div>
+          <p className="mt-0.5 text-xs leading-relaxed">
+            {t("settings.sections.llm.longWritingContextHint")}
+          </p>
+        </div>
+      </div>
+
       {/* Custom Provider Cards - 放在顶部 */}
       {/* Custom Provider Cards - 放在顶部 */}
       <CustomProviderCards />
       <CustomProviderCards />
 
 
-      {/* Built-in Presets - 内置预设,过滤掉通用的"自定义模型"预设 */}
+      {/* Keep every provider; each provider defaults to at least the writing floor. */}
       <div className="space-y-2">
       <div className="space-y-2">
         {LLM_PRESETS.filter((p) => p.id !== "custom").map((preset) => {
         {LLM_PRESETS.filter((p) => p.id !== "custom").map((preset) => {
           const ov = providerConfigs[preset.id]
           const ov = providerConfigs[preset.id]
@@ -158,7 +200,10 @@ function PresetRow({
   const baseUrl = ov.baseUrl ?? preset.baseUrl ?? ""
   const baseUrl = ov.baseUrl ?? preset.baseUrl ?? ""
   const azureApiVersion = ov.azureApiVersion ?? preset.azureApiVersion ?? AZURE_OPENAI_API_VERSION
   const azureApiVersion = ov.azureApiVersion ?? preset.azureApiVersion ?? AZURE_OPENAI_API_VERSION
   const azureModelFamily = ov.azureModelFamily ?? preset.azureModelFamily ?? "auto"
   const azureModelFamily = ov.azureModelFamily ?? preset.azureModelFamily ?? "auto"
-  const context = ov.maxContextSize ?? preset.suggestedContextSize ?? 131072
+  const context = ov.maxContextSize ?? preset.suggestedContextSize ?? MIN_USER_LLM_CONTEXT_SIZE
+  const maxOutputTokens = normalizeUserLlmMaxOutputTokens(
+    ov.maxOutputTokens ?? preset.suggestedMaxOutputTokens,
+  )
   const reasoning = ov.reasoning ?? { mode: "auto" as const }
   const reasoning = ov.reasoning ?? { mode: "auto" as const }
   const localCliIsolation = ov.localCliIsolation === true
   const localCliIsolation = ov.localCliIsolation === true
   const codexCliTimeoutMinutes = Math.max(1, Math.min(240, ov.codexCliTimeoutMinutes ?? 10))
   const codexCliTimeoutMinutes = Math.max(1, Math.min(240, ov.codexCliTimeoutMinutes ?? 10))
@@ -680,9 +725,18 @@ function PresetRow({
             />
             />
           </div>
           </div>
 
 
+          <div className="space-y-2">
+            <Label>{t("settings.sections.llm.maxOutputTokens")}</Label>
+            <OutputTokensSelector
+              value={maxOutputTokens}
+              contextWindow={context}
+              onChange={(v) => onChange({ maxOutputTokens: v })}
+            />
+          </div>
+
           <ReasoningControls
           <ReasoningControls
             value={reasoning}
             value={reasoning}
-            onChange={(reasoning) => onChange({ reasoning })}
+            onChange={(next) => onChange(withOutputRoomForReasoning(next, maxOutputTokens))}
           />
           />
 
 
           <FunctionCallingControls
           <FunctionCallingControls

+ 56 - 7
src/components/sources/outline-chat-panel.tsx

@@ -114,6 +114,14 @@ import {
   resolveUsableModelKey,
   resolveUsableModelKey,
 } from "@/lib/novel/model-resolver";
 } from "@/lib/novel/model-resolver";
 import { hasAvailableModels as hasConfiguredModels } from "@/lib/llm-model-keys";
 import { hasAvailableModels as hasConfiguredModels } from "@/lib/llm-model-keys";
+import {
+  planOutlineRequestBudget,
+  type OutlineBudgetStage,
+} from "@/lib/context-budget";
+import {
+  getEffectiveMaxOutputTokens,
+  thinkingMinMaxTokens,
+} from "@/lib/llm-providers";
 import { ChatModelSelector } from "@/components/chat/chat-model-selector";
 import { ChatModelSelector } from "@/components/chat/chat-model-selector";
 import { highlightCode } from "@/lib/streaming-code-highlight";
 import { highlightCode } from "@/lib/streaming-code-highlight";
 import { separateThinking } from "@/lib/separate-thinking";
 import { separateThinking } from "@/lib/separate-thinking";
@@ -1823,6 +1831,16 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
       // 出错/中断时必须依靠这个变量判断有没有可保留的内容,
       // 出错/中断时必须依靠这个变量判断有没有可保留的内容,
       // 避免整段结果被静默丢弃。
       // 避免整段结果被静默丢弃。
       let bestGeneratedText = "";
       let bestGeneratedText = "";
+      const outlineBudgetStage: OutlineBudgetStage = options.intentPhase === "generation"
+        ? "generation"
+        : "analysis";
+      const outlineRequestBudget = planOutlineRequestBudget({
+        maxContextSize: effectiveLlmConfig.maxContextSize,
+        contextTokenBudget: novelConfig.contextTokenBudget,
+        stage: outlineBudgetStage,
+        maxOutputTokens: getEffectiveMaxOutputTokens(effectiveLlmConfig),
+        thinkingFloorTokens: thinkingMinMaxTokens(effectiveLlmConfig.reasoning ?? { mode: "auto" }),
+      });
 
 
       try {
       try {
         const contextHub = getContextHub(normalizePath(project.path));
         const contextHub = getContextHub(normalizePath(project.path));
@@ -1835,7 +1853,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
           references: tokens.map(describeReferenceForOutlineAgent),
           references: tokens.map(describeReferenceForOutlineAgent),
           messages: historyBeforeSend,
           messages: historyBeforeSend,
           existingSummary: forceRefresh ? undefined : targetConversation?.contextSummary,
           existingSummary: forceRefresh ? undefined : targetConversation?.contextSummary,
-          tokenBudget: novelConfig.contextTokenBudget,
+          tokenBudget: outlineRequestBudget.contextTokenBudget,
           maxContextSize: effectiveLlmConfig.maxContextSize,
           maxContextSize: effectiveLlmConfig.maxContextSize,
           forceRefresh,
           forceRefresh,
         });
         });
@@ -1934,6 +1952,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
         const buildConfigForSkillNames = (
         const buildConfigForSkillNames = (
           skillNames: string[] | undefined,
           skillNames: string[] | undefined,
           disableWriteTools: boolean | undefined,
           disableWriteTools: boolean | undefined,
+          budgetStage: OutlineBudgetStage = outlineBudgetStage,
         ) => {
         ) => {
           const registry = new ToolRegistry();
           const registry = new ToolRegistry();
           const effectiveOutlineWritingSkills = prioritizeOutlineSkills(
           const effectiveOutlineWritingSkills = prioritizeOutlineSkills(
@@ -1980,15 +1999,23 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
                 : {}),
                 : {}),
             },
             },
           );
           );
+          const requestBudget = budgetStage === outlineBudgetStage
+            ? outlineRequestBudget
+            : planOutlineRequestBudget({
+                maxContextSize: effectiveLlmConfig.maxContextSize,
+                contextTokenBudget: novelConfig.contextTokenBudget,
+                stage: budgetStage,
+                maxOutputTokens: getEffectiveMaxOutputTokens(effectiveLlmConfig),
+                thinkingFloorTokens: thinkingMinMaxTokens(
+                  effectiveLlmConfig.reasoning ?? { mode: "auto" },
+                ),
+              });
           return {
           return {
             agentConfig: {
             agentConfig: {
               ...agentConfig,
               ...agentConfig,
               requestOverrides: {
               requestOverrides: {
                 ...agentConfig.requestOverrides,
                 ...agentConfig.requestOverrides,
-                max_tokens: Math.max(
-                  agentConfig.requestOverrides?.max_tokens ?? 0,
-                  32768,
-                ),
+                max_tokens: requestBudget.outputTokens,
                 userMemorySurface: "ai-outline" as const,
                 userMemorySurface: "ai-outline" as const,
                 userMemoryProjectKey: normalizePath(project.path),
                 userMemoryProjectKey: normalizePath(project.path),
                 userMemorySessionKey: capturedConvId,
                 userMemorySessionKey: capturedConvId,
@@ -2005,11 +2032,13 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
             disableWriteTools?: boolean;
             disableWriteTools?: boolean;
             streamToUser?: boolean;
             streamToUser?: boolean;
             statusText?: string;
             statusText?: string;
+            budgetStage?: OutlineBudgetStage;
           } = {},
           } = {},
         ): Promise<{ text: string; record: AgentRunRecord; error?: Error; reasoning_content: string }> => {
         ): Promise<{ text: string; record: AgentRunRecord; error?: Error; reasoning_content: string }> => {
           const { agentConfig, registry } = buildConfigForSkillNames(
           const { agentConfig, registry } = buildConfigForSkillNames(
             optionsForRun.skillNames,
             optionsForRun.skillNames,
             optionsForRun.disableWriteTools,
             optionsForRun.disableWriteTools,
+            optionsForRun.budgetStage,
           );
           );
           let runText = "";
           let runText = "";
           let runReasoningContent = "";
           let runReasoningContent = "";
@@ -2167,6 +2196,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
                   skillNames: options.preferredSkillNames,
                   skillNames: options.preferredSkillNames,
                   disableWriteTools: true,
                   disableWriteTools: true,
                   streamToUser: false,
                   streamToUser: false,
+                  budgetStage: "analysis",
                 });
                 });
                 if (charRun.error) {
                 if (charRun.error) {
                   throw new Error(charRun.error.message);
                   throw new Error(charRun.error.message);
@@ -2239,6 +2269,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
               skillNames: [],
               skillNames: [],
               disableWriteTools: true,
               disableWriteTools: true,
               statusText: "正在动态规划 Agent 任务…",
               statusText: "正在动态规划 Agent 任务…",
+              budgetStage: "analysis",
             });
             });
             const dynamicPlan = parseDynamicOutlinePlan(
             const dynamicPlan = parseDynamicOutlinePlan(
               plannerRun.text,
               plannerRun.text,
@@ -2852,6 +2883,13 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
         toast.error("请先在设置中配置并选择一个可用的 AI 模型。");
         toast.error("请先在设置中配置并选择一个可用的 AI 模型。");
         return;
         return;
       }
       }
+      const resumeRequestBudget = planOutlineRequestBudget({
+        maxContextSize: effectiveLlmConfig.maxContextSize,
+        contextTokenBudget: novelConfig.contextTokenBudget,
+        stage: "generation",
+        maxOutputTokens: getEffectiveMaxOutputTokens(effectiveLlmConfig),
+        thinkingFloorTokens: thinkingMinMaxTokens(effectiveLlmConfig.reasoning ?? { mode: "auto" }),
+      });
 
 
       const capturedConvId = activeConversationId;
       const capturedConvId = activeConversationId;
       const runId = crypto.randomUUID();
       const runId = crypto.randomUUID();
@@ -2880,7 +2918,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
               content: message.content,
               content: message.content,
             })),
             })),
             existingSummary: conv.contextSummary,
             existingSummary: conv.contextSummary,
-            tokenBudget: novelConfig.contextTokenBudget,
+            tokenBudget: resumeRequestBudget.contextTokenBudget,
             maxContextSize: effectiveLlmConfig.maxContextSize,
             maxContextSize: effectiveLlmConfig.maxContextSize,
           });
           });
           if (contextHubResult) {
           if (contextHubResult) {
@@ -2953,6 +2991,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
               ...c,
               ...c,
               requestOverrides: {
               requestOverrides: {
                 ...c.requestOverrides,
                 ...c.requestOverrides,
+                max_tokens: resumeRequestBudget.outputTokens,
                 userMemorySurface: "ai-outline" as const,
                 userMemorySurface: "ai-outline" as const,
                 userMemoryProjectKey: normalizePath(project.path),
                 userMemoryProjectKey: normalizePath(project.path),
                 userMemorySessionKey: capturedConvId,
                 userMemorySessionKey: capturedConvId,
@@ -3249,6 +3288,15 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
         const historyMessages = regenerationInput.history satisfies AgentMessage[];
         const historyMessages = regenerationInput.history satisfies AgentMessage[];
         let contextHubSnapshot: ContextHubSnapshotRef | undefined;
         let contextHubSnapshot: ContextHubSnapshotRef | undefined;
         let contextHubResult: ContextHubResult | null = null;
         let contextHubResult: ContextHubResult | null = null;
+        const regenerationRequestBudget = planOutlineRequestBudget({
+          maxContextSize: effectiveLlmConfig.maxContextSize,
+          contextTokenBudget: novelConfig.contextTokenBudget,
+          stage: "generation",
+          maxOutputTokens: getEffectiveMaxOutputTokens(effectiveLlmConfig),
+          thinkingFloorTokens: thinkingMinMaxTokens(
+            effectiveLlmConfig.reasoning ?? { mode: "auto" },
+          ),
+        });
         try {
         try {
           const contextHub = getContextHub(normalizePath(project.path));
           const contextHub = getContextHub(normalizePath(project.path));
           contextHubResult = await contextHub.prepare({
           contextHubResult = await contextHub.prepare({
@@ -3259,7 +3307,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
             intent: "generate",
             intent: "generate",
             messages: historyMessages,
             messages: historyMessages,
             existingSummary: undefined,
             existingSummary: undefined,
-            tokenBudget: novelConfig.contextTokenBudget,
+            tokenBudget: regenerationRequestBudget.contextTokenBudget,
             maxContextSize: effectiveLlmConfig.maxContextSize,
             maxContextSize: effectiveLlmConfig.maxContextSize,
           });
           });
           if (contextHubResult && isCurrentRun()) {
           if (contextHubResult && isCurrentRun()) {
@@ -3340,6 +3388,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
         );
         );
         agentConfig.requestOverrides = {
         agentConfig.requestOverrides = {
           ...agentConfig.requestOverrides,
           ...agentConfig.requestOverrides,
+          max_tokens: regenerationRequestBudget.outputTokens,
           userMemorySurface: "ai-outline",
           userMemorySurface: "ai-outline",
           userMemoryProjectKey: normalizePath(project.path),
           userMemoryProjectKey: normalizePath(project.path),
           userMemorySessionKey: capturedConvId,
           userMemorySessionKey: capturedConvId,

+ 9 - 1
src/i18n/en.json

@@ -760,6 +760,8 @@
       "llm": {
       "llm": {
         "title": "Large Language / LLM Model",
         "title": "Large Language / LLM Model",
         "description": "Each provider has an independent configuration. Turning one on makes it the active provider and turns the others off. Edits save immediately, and API keys remain separate per provider.",
         "description": "Each provider has an independent configuration. Turning one on makes it the active provider and turns the others off. Edits save immediately, and API keys remain separate per provider.",
+        "longWritingContextTitle": "Chapter and outline writing requires at least a 200K context window",
+        "longWritingContextHint": "The minimum is 204800 tokens (about 200K). Every built-in provider defaults to at least this value. For custom or manually entered models, confirm that the server actually supports that window.",
         "expand": "Expand configuration",
         "expand": "Expand configuration",
         "collapse": "Collapse",
         "collapse": "Collapse",
         "toggleOff": "Disable this provider",
         "toggleOff": "Disable this provider",
@@ -771,6 +773,12 @@
         "apiKey": "API Key",
         "apiKey": "API Key",
         "model": "Model",
         "model": "Model",
         "contextWindow": "Context Window",
         "contextWindow": "Context Window",
+        "contextWindowValue": "{{value}} tokens",
+        "contextWindowHint": "The context length from the model's spec sheet, in tokens. Minimum 200K — older configs below that are raised automatically on startup.",
+        "maxOutputTokens": "Output Limit",
+        "maxOutputTokensValue": "{{value}} tokens",
+        "maxOutputTokensHint": "How many tokens this model can emit in one reply, used to avoid rejected requests. It declares a capability rather than a request size: each task sends the smaller of what it needs and this limit, so raising it does not make replies longer.",
+        "maxOutputTokensExceedsWindow": "Exceeds the context window; it will be reduced to what the window can hold",
         "wireOpenAi": "OpenAI Compatible",
         "wireOpenAi": "OpenAI Compatible",
         "wireResponses": "Responses API",
         "wireResponses": "Responses API",
         "wireAnthropic": "Anthropic Compatible",
         "wireAnthropic": "Anthropic Compatible",
@@ -1335,7 +1343,7 @@
       "searchTopKHint": "Controls how many relevant memory records are retrieved and injected into the writing context. Higher values add more references, but can add noise and consume context.",
       "searchTopKHint": "Controls how many relevant memory records are retrieved and injected into the writing context. Higher values add more references, but can add noise and consume context.",
       "contextTokenBudget": "Context Token Budget",
       "contextTokenBudget": "Context Token Budget",
       "contextTokenBudgetHint": "0 = auto from model window",
       "contextTokenBudgetHint": "0 = auto from model window",
-      "contextTokenBudgetHelp": "Limits context tokens injected into one novel writing or extraction call. 0 means no extra limit: the budget is derived from a safe fraction of the model context window. Deep chapter writing first reserves output headroom as max(2× chapter target at CJK density, chapter maxOutputTokens), then allocates the context pack.",
+      "contextTokenBudgetHelp": "Limits context tokens injected into one novel writing or extraction call. 0 derives the budget from the model window. Chapter and outline workflows reserve system prompts, tools, and the actual output limit before allocating the remaining budget to the context pack.",
       "chatHistoryLength": "Chat History Length",
       "chatHistoryLength": "Chat History Length",
       "chatHistoryLengthHint": "Number of previous messages sent to the AI with each request. More history gives fuller context but uses more tokens.",
       "chatHistoryLengthHint": "Number of previous messages sent to the AI with each request. More history gives fuller context but uses more tokens.",
       "chatHistoryLengthHelp": "Controls how many AI chat messages are included in each request. Higher values preserve more conversation context and consume more tokens.",
       "chatHistoryLengthHelp": "Controls how many AI chat messages are included in each request. Higher values preserve more conversation context and consume more tokens.",

+ 9 - 1
src/i18n/zh.json

@@ -467,6 +467,8 @@
       "llm": {
       "llm": {
         "title": "大语言/LLM模型",
         "title": "大语言/LLM模型",
         "description": "配置不同大语言模型服务商的 API Key、模型和连接方式。自定义模型保留在最上方,内置预设模型在下方可单独启用。",
         "description": "配置不同大语言模型服务商的 API Key、模型和连接方式。自定义模型保留在最上方,内置预设模型在下方可单独启用。",
+        "longWritingContextTitle": "正文和大纲写作要求至少 200K 上下文",
+        "longWritingContextHint": "最低值为 204800 tokens(约 200K)。所有内置供应商默认值均不低于该值。自定义或手动输入的模型请确认服务端真实支持该窗口。",
         "expand": "展开配置",
         "expand": "展开配置",
         "collapse": "收起配置",
         "collapse": "收起配置",
         "toggleOff": "停用此模型",
         "toggleOff": "停用此模型",
@@ -478,6 +480,12 @@
         "apiKey": "API 密钥",
         "apiKey": "API 密钥",
         "model": "模型",
         "model": "模型",
         "contextWindow": "上下文窗口",
         "contextWindow": "上下文窗口",
+        "contextWindowValue": "{{value}} tokens",
+        "contextWindowHint": "模型规格表上的上下文长度,单位为 token。最低 200K,低于该值的旧配置会在启动时自动提升。",
+        "maxOutputTokens": "输出上限",
+        "maxOutputTokensValue": "{{value}} tokens",
+        "maxOutputTokensHint": "该模型单次回复最多能输出多少 token,用于避免请求被供应商拒绝。这是能力声明而非请求长度:实际发出的取「本次任务所需」与该上限的较小值,调高不会让回复变长。",
+        "maxOutputTokensExceedsWindow": "超出上下文窗口,实际会被压回窗口可容纳的范围",
         "wireOpenAi": "OpenAI 兼容",
         "wireOpenAi": "OpenAI 兼容",
         "wireResponses": "Responses API",
         "wireResponses": "Responses API",
         "wireAnthropic": "Anthropic 兼容",
         "wireAnthropic": "Anthropic 兼容",
@@ -1204,7 +1212,7 @@
       "searchTopKHint": "控制从小说记忆库中检索并注入上下文的相关资料条数。数值越大,参考资料越多,但噪音和上下文占用也会增加。",
       "searchTopKHint": "控制从小说记忆库中检索并注入上下文的相关资料条数。数值越大,参考资料越多,但噪音和上下文占用也会增加。",
       "contextTokenBudget": "上下文 Token 预算",
       "contextTokenBudget": "上下文 Token 预算",
       "contextTokenBudgetHint": "0 表示按模型窗口自动计算",
       "contextTokenBudgetHint": "0 表示按模型窗口自动计算",
-      "contextTokenBudgetHelp": "限制一次小说写作或资料提取时可注入的上下文 Token 数量,避免提示词过长。设置为 0 时不额外限制,按模型上下文窗口的安全比例自动计算。深度写作会先按「单章目标字数×2(按中文密度折算)与章节 maxOutputTokens 取较大值」预留正文输出,再分配资料包。",
+      "contextTokenBudgetHelp": "限制一次小说写作或资料提取时可注入的上下文 Token 数量。设置为 0 时按模型窗口自动计算。正文和大纲工作流会统一预留系统提示、工具和实际输出空间,再把剩余预算分配给资料包。",
       "chatHistoryLength": "对话历史长度",
       "chatHistoryLength": "对话历史长度",
       "chatHistoryLengthHint": "每次请求发给 AI 的历史消息条数。多 = 上下文更完整但更费 token。",
       "chatHistoryLengthHint": "每次请求发给 AI 的历史消息条数。多 = 上下文更完整但更费 token。",
       "chatHistoryLengthHelp": "控制 AI 会话中每次请求携带多少条历史消息。数量越多,AI 记得的上下文越完整,但消耗的 token 也越多。",
       "chatHistoryLengthHelp": "控制 AI 会话中每次请求携带多少条历史消息。数量越多,AI 记得的上下文越完整,但消耗的 token 也越多。",

+ 16 - 2
src/lib/agent/runner.ts

@@ -16,7 +16,7 @@ import {
 import { getEffectiveMaxContextSize, type ChatMessage } from "../llm-providers"
 import { getEffectiveMaxContextSize, type ChatMessage } from "../llm-providers"
 import { isReasoningDisabled, isReasoningOnlyResponseError, withReasoningDisabled } from "../reasoning-retry"
 import { isReasoningDisabled, isReasoningOnlyResponseError, withReasoningDisabled } from "../reasoning-retry"
 import { addLlmUsage } from "../llm-usage"
 import { addLlmUsage } from "../llm-usage"
-import { trimChatMessagesToBudget } from "../chat-request-budget"
+import { trimChatMessagesToTokenBudget } from "../chat-request-budget"
 import { logReasoningReplay } from "../reasoning-replay-debug"
 import { logReasoningReplay } from "../reasoning-replay-debug"
 import { ToolEvidenceLedger } from "./tool-evidence-ledger"
 import { ToolEvidenceLedger } from "./tool-evidence-ledger"
 import {
 import {
@@ -168,9 +168,23 @@ export class AgentRunner {
         return record
         return record
       }
       }
       const streamRound = async () => {
       const streamRound = async () => {
+        // maxContextSize is already a token count; the remaining quarter of the
+        // window covers the response and prompt scaffolding.
         const effectiveContext = getEffectiveMaxContextSize(config.llmConfig)
         const effectiveContext = getEffectiveMaxContextSize(config.llmConfig)
         const internalBudget = Math.max(1, Math.floor(effectiveContext * 0.75))
         const internalBudget = Math.max(1, Math.floor(effectiveContext * 0.75))
-        const compacted = trimChatMessagesToBudget(workingMessages as ChatMessage[], internalBudget) as AgentMessage[]
+        let compacted: AgentMessage[]
+        try {
+          compacted = trimChatMessagesToTokenBudget(
+            workingMessages as ChatMessage[],
+            internalBudget,
+          ) as AgentMessage[]
+        } catch {
+          // streamChat retries with a 512-token output floor before giving up;
+          // surface a readable reason instead of the bare budget error.
+          throw new Error(
+            "模型上下文不足:当前对话即使压缩后仍放不下系统提示与最新请求。请缩短输入,或在设置中调高该模型的上下文窗口。",
+          )
+        }
         workingMessages.splice(0, workingMessages.length, ...compacted)
         workingMessages.splice(0, workingMessages.length, ...compacted)
         await streamChat(
         await streamChat(
           config.llmConfig,
           config.llmConfig,

+ 2 - 2
src/lib/agent/tools/index.ts

@@ -37,7 +37,7 @@ export interface VirtualToolContext {
   contextPack?: ContextPack
   contextPack?: ContextPack
   /** ContextPack token budget; resolved from model window when omitted. */
   /** ContextPack token budget; resolved from model window when omitted. */
   tokenBudget?: number
   tokenBudget?: number
-  /** Session model context window in characters. */
+  /** Session model context window in tokens. */
   maxContextSize?: number
   maxContextSize?: number
 }
 }
 
 
@@ -52,7 +52,7 @@ export interface ToolFactoryOptions {
   mcpTools?: Tool[]
   mcpTools?: Tool[]
   draftMode?: boolean
   draftMode?: boolean
   projectPath?: string
   projectPath?: string
-  /** Session model context window in characters (for trim_context defaults). */
+  /** Session model context window in tokens (for trim_context defaults). */
   maxContextSize?: number
   maxContextSize?: number
   sourceConversationId?: string
   sourceConversationId?: string
   sourceMessageId?: string
   sourceMessageId?: string

+ 83 - 1
src/lib/chat-request-budget.test.ts

@@ -1,6 +1,12 @@
 import { describe, expect, it } from "vitest"
 import { describe, expect, it } from "vitest"
 import type { ChatMessage } from "./llm-client"
 import type { ChatMessage } from "./llm-client"
-import { trimChatMessagesToBudget } from "./chat-request-budget"
+import {
+  estimateChatMessagesTokens,
+  estimateRequestScaffoldTokens,
+  trimChatMessagesToBudget,
+  trimChatMessagesToTokenBudget,
+} from "./chat-request-budget"
+import { LlmContextBudgetError } from "./context-budget"
 
 
 function text(length: number, char = "x"): string {
 function text(length: number, char = "x"): string {
   return char.repeat(length)
   return char.repeat(length)
@@ -117,3 +123,79 @@ describe("trimChatMessagesToBudget", () => {
     expect(trimmed.at(-1)?.content).toBe("现在只回答新的问题")
     expect(trimmed.at(-1)?.content).toBe("现在只回答新的问题")
   })
   })
 })
 })
+
+describe("trimChatMessagesToTokenBudget", () => {
+  it("uses CJK-aware estimation for Chinese, English, and mixed content", () => {
+    const chinese = [{ role: "user" as const, content: "中".repeat(100) }]
+    const english = [{ role: "user" as const, content: "a".repeat(100) }]
+    const mixed = [{ role: "user" as const, content: `${"中".repeat(50)}${"a".repeat(100)}` }]
+    expect(estimateChatMessagesTokens(chinese)).toBeGreaterThan(estimateChatMessagesTokens(english))
+    expect(estimateChatMessagesTokens(mixed)).toBeGreaterThan(estimateChatMessagesTokens(english))
+    expect(estimateChatMessagesTokens(mixed)).toBeLessThan(estimateChatMessagesTokens(chinese))
+  })
+
+  it("counts tool schema as request scaffold", () => {
+    const small = estimateRequestScaffoldTokens([{ type: "function", function: { name: "read" } }])
+    const large = estimateRequestScaffoldTokens([{
+      type: "function",
+      function: {
+        name: "read",
+        description: "读取资料".repeat(100),
+        parameters: { type: "object", properties: { path: { type: "string" } } },
+      },
+    }])
+    expect(small).toBeGreaterThan(0)
+    expect(large).toBeGreaterThan(small)
+  })
+
+  it("drops old tool protocol groups and preserves non-empty system/current user content", () => {
+    const messages: ChatMessage[] = [
+      { role: "system", content: `系统约束:${"规则".repeat(500)}` },
+      { role: "assistant", content: "", tool_calls: [{ id: "old", type: "function", function: { name: "read", arguments: "{}" } }] },
+      { role: "tool", content: "旧结果".repeat(1_000), tool_call_id: "old", name: "read" },
+      { role: "user", content: `任务目标:${"正文".repeat(1_000)}结尾限制:保持人物关系。` },
+    ]
+    const trimmed = trimChatMessagesToTokenBudget(messages, 500)
+    expect(estimateChatMessagesTokens(trimmed)).toBeLessThanOrEqual(500)
+    expect(trimmed.some((message) => message.tool_call_id === "old")).toBe(false)
+    expect(String(trimmed[0]?.content).trim()).not.toBe("")
+    expect(String(trimmed.at(-1)?.content)).toContain("任务目标")
+    expect(String(trimmed.at(-1)?.content)).toContain("保持人物关系")
+  })
+
+  it("fails explicitly instead of emptying protected messages", () => {
+    expect(() => trimChatMessagesToTokenBudget([
+      { role: "system", content: "必须遵守约束" },
+      { role: "user", content: "生成第一卷完整大纲" },
+    ], 5)).toThrow(LlmContextBudgetError)
+  })
+
+  it("compresses the same input whether or not a mid-conversation system exists", () => {
+    // AgentRunner pushes a required-tool system message into the middle of the
+    // history. It is droppable history, so validating the survivors against the
+    // original system list by position misaligned and rejected a trim that had
+    // actually succeeded.
+    const leading: ChatMessage = { role: "system", content: `系统约束:${"规则".repeat(500)}` }
+    const history: ChatMessage[] = [
+      { role: "user", content: "早前请求".repeat(200) },
+      { role: "assistant", content: "早前回复".repeat(200) },
+    ]
+    const current: ChatMessage = {
+      role: "user",
+      content: `任务目标:${"正文".repeat(1_000)}结尾限制:保持人物关系。`,
+    }
+    const withoutMidSystem = trimChatMessagesToTokenBudget([leading, ...history, current], 500)
+    const withMidSystem = trimChatMessagesToTokenBudget([
+      leading,
+      ...history,
+      { role: "system", content: "本轮必须调用 read_chapter。" },
+      current,
+    ], 500)
+
+    expect(estimateChatMessagesTokens(withMidSystem)).toBeLessThanOrEqual(500)
+    expect(String(withMidSystem[0]?.content).trim()).not.toBe("")
+    expect(String(withMidSystem.at(-1)?.content)).toContain("任务目标")
+    expect(withMidSystem.map((message) => message.role))
+      .toEqual(withoutMidSystem.map((message) => message.role))
+  })
+})

+ 270 - 0
src/lib/chat-request-budget.ts

@@ -1,4 +1,6 @@
 import type { ChatMessage, ContentBlock } from "./llm-providers"
 import type { ChatMessage, ContentBlock } from "./llm-providers"
+import { LlmContextBudgetError } from "./context-budget"
+import { estimateContextTokens } from "./context-hub/token-estimator"
 
 
 const HISTORY_TRUNCATED_MARKER = "[history truncated]\n"
 const HISTORY_TRUNCATED_MARKER = "[history truncated]\n"
 const CONTENT_TRUNCATED_MARKER = "\n[内容已压缩,保留首尾]\n"
 const CONTENT_TRUNCATED_MARKER = "\n[内容已压缩,保留首尾]\n"
@@ -121,6 +123,274 @@ function groupHistory(messages: ChatMessage[]): ChatMessage[][] {
   return groups
   return groups
 }
 }
 
 
+const MESSAGE_OVERHEAD_TOKENS = 4
+
+function contentTokenLength(content: ChatMessage["content"]): number {
+  if (typeof content === "string") return estimateContextTokens(content)
+  return content.reduce((sum, block) => {
+    if (block.type === "text") return sum + estimateContextTokens(block.text)
+    return sum + estimateContextTokens(block.dataBase64)
+  }, 0)
+}
+
+function toolMetadataTokens(message: ChatMessage): number {
+  const toolCallTokens = message.tool_calls?.reduce(
+    (sum, call) => sum + estimateContextTokens(
+      `${call.id}\n${call.function.name}\n${call.function.arguments}`,
+    ),
+    0,
+  ) ?? 0
+  return toolCallTokens + estimateContextTokens(
+    `${message.tool_call_id ?? ""}\n${message.name ?? ""}`,
+  )
+}
+
+function messageTokenLength(message: ChatMessage): number {
+  return MESSAGE_OVERHEAD_TOKENS + contentTokenLength(message.content) + toolMetadataTokens(message)
+}
+
+function minimumMessageTokenLength(message: ChatMessage, requireContent: boolean): number {
+  const fixedMetadataTokens = MESSAGE_OVERHEAD_TOKENS + estimateContextTokens(
+    `${message.tool_call_id ?? ""}\n${message.name ?? ""}`,
+  )
+  const minimumToolTokens = message.tool_calls?.reduce(
+    (sum, call) => sum + estimateContextTokens(`${call.id}\n${call.function.name}\n{}`),
+    0,
+  ) ?? 0
+  return fixedMetadataTokens + minimumToolTokens + (requireContent ? 1 : 0)
+}
+
+export function estimateChatMessagesTokens(messages: ChatMessage[]): number {
+  return messages.reduce((sum, message) => sum + messageTokenLength(message), 0)
+}
+
+export function estimateRequestScaffoldTokens(tools: unknown): number {
+  if (!tools) return 0
+  try {
+    return estimateContextTokens(JSON.stringify(tools))
+  } catch {
+    return 0
+  }
+}
+
+function clampTextToTokenBudget(
+  text: string,
+  maxTokens: number,
+  preserveHead: boolean,
+): string {
+  if (estimateContextTokens(text) <= maxTokens) return text
+  if (maxTokens <= 0) return ""
+  let low = 0
+  let high = text.length
+  let best = ""
+  while (low <= high) {
+    const mid = Math.floor((low + high) / 2)
+    const candidate = preserveHead
+      ? clampHeadTail(text, mid)
+      : clampTail(text, mid)
+    if (estimateContextTokens(candidate) <= maxTokens) {
+      best = candidate
+      low = mid + 1
+    } else {
+      high = mid - 1
+    }
+  }
+  return best
+}
+
+function trimContentToTokenBudget(
+  content: ChatMessage["content"],
+  maxTokens: number,
+  preserveHead = false,
+): ChatMessage["content"] {
+  if (typeof content === "string") {
+    return clampTextToTokenBudget(content, maxTokens, preserveHead)
+  }
+  let remaining = maxTokens
+  const reversed: ContentBlock[] = []
+  for (let index = content.length - 1; index >= 0; index -= 1) {
+    const block = content[index]
+    if (!block) continue
+    if (block.type !== "text") {
+      const tokens = estimateContextTokens(block.dataBase64)
+      if (tokens <= remaining) {
+        reversed.push(block)
+        remaining -= tokens
+      }
+      continue
+    }
+    const text = clampTextToTokenBudget(block.text, remaining, preserveHead)
+    if (text) {
+      reversed.push({ ...block, text })
+      remaining -= estimateContextTokens(text)
+    }
+    if (remaining <= 0) break
+  }
+  return reversed.reverse()
+}
+
+function trimMessageToTokenBudget(
+  message: ChatMessage,
+  maxTokens: number,
+  preserveHead = false,
+): ChatMessage {
+  const fixedMetadataTokens = MESSAGE_OVERHEAD_TOKENS + estimateContextTokens(
+    `${message.tool_call_id ?? ""}\n${message.name ?? ""}`,
+  )
+  const originalToolCalls = message.tool_calls
+  const minimumToolTokens = originalToolCalls?.reduce(
+    (sum, call) => sum + estimateContextTokens(`${call.id}\n${call.function.name}\n{}`),
+    0,
+  ) ?? 0
+  const contentBudget = Math.max(0, maxTokens - fixedMetadataTokens - minimumToolTokens)
+  const content = trimContentToTokenBudget(message.content, contentBudget, preserveHead)
+  let remaining = Math.max(
+    0,
+    maxTokens - fixedMetadataTokens - contentTokenLength(content),
+  )
+  const toolCalls = originalToolCalls?.map((call) => {
+    const fixed = estimateContextTokens(`${call.id}\n${call.function.name}\n`)
+    const fullArguments = estimateContextTokens(call.function.arguments)
+    const argumentsValue = fixed + fullArguments <= remaining
+      ? call.function.arguments
+      : "{}"
+    remaining = Math.max(
+      0,
+      remaining - fixed - estimateContextTokens(argumentsValue),
+    )
+    return {
+      ...call,
+      function: { ...call.function, arguments: argumentsValue },
+    }
+  })
+  return {
+    ...message,
+    content,
+    ...(toolCalls ? { tool_calls: toolCalls } : {}),
+  }
+}
+
+function hasNonEmptyContent(message: ChatMessage | undefined): boolean {
+  if (!message) return false
+  if (typeof message.content === "string") return message.content.trim().length > 0
+  return message.content.some((block) => block.type !== "text" || block.text.trim().length > 0)
+}
+
+/**
+ * Token-aware request trimmer. Tool-call/result groups stay paired; system
+ * constraints and the latest user request may be shortened but never emptied.
+ */
+export function trimChatMessagesToTokenBudget(
+  messages: ChatMessage[],
+  maxTokens: number,
+): ChatMessage[] {
+  if (messages.length === 0) return messages
+  if (!Number.isFinite(maxTokens) || maxTokens <= 0) throw new LlmContextBudgetError()
+  if (estimateChatMessagesTokens(messages) <= maxTokens) return messages
+
+  const leadingSystems: ChatMessage[] = []
+  let firstNonSystem = 0
+  while (firstNonSystem < messages.length && messages[firstNonSystem]?.role === "system") {
+    leadingSystems.push(messages[firstNonSystem]!)
+    firstNonSystem += 1
+  }
+  const bodyGroups = groupHistory(messages.slice(firstNonSystem))
+  let latestUserGroup = -1
+  for (let index = bodyGroups.length - 1; index >= 0; index -= 1) {
+    if (bodyGroups[index]!.some((message) => message.role === "user")) {
+      latestUserGroup = index
+      break
+    }
+  }
+  const retainedGroups = bodyGroups.map((group, index) => ({
+    group,
+    protected: index === latestUserGroup || index === bodyGroups.length - 1,
+  }))
+  let next = [...leadingSystems, ...retainedGroups.flatMap((entry) => entry.group)]
+
+  while (estimateChatMessagesTokens(next) > maxTokens) {
+    const removableIndex = retainedGroups.findIndex((entry) => !entry.protected)
+    if (removableIndex < 0) break
+    retainedGroups.splice(removableIndex, 1)
+    next = [...leadingSystems, ...retainedGroups.flatMap((entry) => entry.group)]
+  }
+  if (estimateChatMessagesTokens(next) <= maxTokens) return next
+
+  // Systems that survived group-dropping. Mid-conversation system messages
+  // (e.g. the required-tool nudge pushed by AgentRunner) are ordinary history
+  // and may legitimately be gone by now; only what is still here has to stay
+  // non-empty through compression. Comparing against the original list by
+  // position instead would misalign the moment any system is dropped, and
+  // report a budget failure for a trim that actually succeeded.
+  // Compression below replaces entries in place, so these indices stay valid.
+  const protectedSystemIndices = next.reduce<number[]>((indices, message, index) => {
+    if (message.role === "system" && hasNonEmptyContent(message)) indices.push(index)
+    return indices
+  }, [])
+
+  let latestUserIndex = -1
+  for (let index = next.length - 1; index >= 0; index -= 1) {
+    if (next[index]?.role === "user") {
+      latestUserIndex = index
+      break
+    }
+  }
+
+  // Old assistant/tool payloads are expendable before protected instructions.
+  for (let index = 0; index < next.length && estimateChatMessagesTokens(next) > maxTokens; index += 1) {
+    if (index === latestUserIndex || next[index]?.role === "system") continue
+    const current = next[index]!
+    const excess = estimateChatMessagesTokens(next) - maxTokens
+    next[index] = trimMessageToTokenBudget(
+      current,
+      Math.max(
+        minimumMessageTokenLength(current, false),
+        messageTokenLength(current) - excess,
+      ),
+    )
+  }
+
+  // System messages remain non-empty and retain both ends when compressed.
+  for (let index = 0; index < next.length && estimateChatMessagesTokens(next) > maxTokens; index += 1) {
+    if (next[index]?.role !== "system") continue
+    const current = next[index]!
+    const excess = estimateChatMessagesTokens(next) - maxTokens
+    next[index] = trimMessageToTokenBudget(
+      current,
+      Math.max(
+        minimumMessageTokenLength(current, true),
+        messageTokenLength(current) - excess,
+      ),
+      true,
+    )
+  }
+
+  if (latestUserIndex >= 0 && estimateChatMessagesTokens(next) > maxTokens) {
+    const current = next[latestUserIndex]!
+    const excess = estimateChatMessagesTokens(next) - maxTokens
+    next[latestUserIndex] = trimMessageToTokenBudget(
+      current,
+      Math.max(
+        minimumMessageTokenLength(current, true),
+        messageTokenLength(current) - excess,
+      ),
+      true,
+    )
+  }
+
+  const protectedSystemsValid = protectedSystemIndices
+    .every((index) => hasNonEmptyContent(next[index]))
+  const latestUserValid = latestUserIndex < 0 || hasNonEmptyContent(next[latestUserIndex])
+  if (
+    estimateChatMessagesTokens(next) > maxTokens
+    || !protectedSystemsValid
+    || !latestUserValid
+  ) {
+    throw new LlmContextBudgetError()
+  }
+  return next
+}
+
 /**
 /**
  * Trims packed chat messages by character budget before sending them to an LLM.
  * Trims packed chat messages by character budget before sending them to an LLM.
  * The current user request is preserved because it carries the user's latest intent.
  * The current user request is preserved because it carries the user's latest intent.

+ 16 - 22
src/lib/context-budget.contract.spec.ts

@@ -2,11 +2,10 @@ import { readFileSync } from "node:fs"
 import { resolve } from "node:path"
 import { resolve } from "node:path"
 import { describe, expect, it } from "vitest"
 import { describe, expect, it } from "vitest"
 import {
 import {
-  computeContextBudget,
   computeNovelContextTokenBudget,
   computeNovelContextTokenBudget,
   computeWritingContextPackTokenBudget,
   computeWritingContextPackTokenBudget,
+  planChapterRequestBudget,
   resolveContextPackTokenBudget,
   resolveContextPackTokenBudget,
-  WRITING_OUTPUT_RESERVE_MULTIPLIER,
 } from "./context-budget"
 } from "./context-budget"
 
 
 describe("context pack budget contracts", () => {
 describe("context pack budget contracts", () => {
@@ -15,50 +14,44 @@ describe("context pack budget contracts", () => {
       const general = resolveContextPackTokenBudget({
       const general = resolveContextPackTokenBudget({
         maxContextSize,
         maxContextSize,
         contextTokenBudget: 0,
         contextTokenBudget: 0,
-        langScale: 1,
       })
       })
       const writing = computeWritingContextPackTokenBudget({
       const writing = computeWritingContextPackTokenBudget({
         maxContextSize,
         maxContextSize,
         contextTokenBudget: 0,
         contextTokenBudget: 0,
         chapterTargetChars: 3_000,
         chapterTargetChars: 3_000,
-        langScale: 1,
       })
       })
       expect(Number.isFinite(general)).toBe(true)
       expect(Number.isFinite(general)).toBe(true)
       expect(Number.isFinite(writing)).toBe(true)
       expect(Number.isFinite(writing)).toBe(true)
       expect(general).toBeGreaterThan(0)
       expect(general).toBeGreaterThan(0)
       expect(writing).toBeGreaterThan(0)
       expect(writing).toBeGreaterThan(0)
-      expect(general).toBeLessThanOrEqual(computeNovelContextTokenBudget(maxContextSize, 0, 1))
-      expect(writing).toBeLessThanOrEqual(general)
+      expect(general).toBeLessThanOrEqual(computeNovelContextTokenBudget(maxContextSize, 0))
+      const normalizedGeneral = resolveContextPackTokenBudget({
+        maxContextSize: Math.max(204_800, maxContextSize),
+        contextTokenBudget: 0,
+      })
+      expect(writing).toBeLessThanOrEqual(normalizedGeneral)
     }
     }
   })
   })
 
 
-  it("writing budget collapses to zero when the window cannot fit output reserve", () => {
+  it("writing adapter migrates a legacy small window before allocating", () => {
     const writing = computeWritingContextPackTokenBudget({
     const writing = computeWritingContextPackTokenBudget({
       maxContextSize: 32_000,
       maxContextSize: 32_000,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })
     })
-    expect(writing).toBe(0)
+    expect(writing).toBe(133_120)
   })
   })
 
 
   it("writing pack leaves room for output-token reserve plus scaffold", () => {
   it("writing pack leaves room for output-token reserve plus scaffold", () => {
     for (const chapterTargetChars of [2_000, 3_000, 6_000]) {
     for (const chapterTargetChars of [2_000, 3_000, 6_000]) {
       for (const maxContextSize of [64_000, 204_800]) {
       for (const maxContextSize of [64_000, 204_800]) {
-        const { maxCtx } = computeContextBudget(maxContextSize, 1)
-        const packTokens = computeWritingContextPackTokenBudget({
+        const plan = planChapterRequestBudget({
           maxContextSize,
           maxContextSize,
           chapterTargetChars,
           chapterTargetChars,
-          langScale: 1,
+          stage: "generation",
         })
         })
-        const maxOutputTokens = chapterTargetChars === 3_000
-          ? 8_000
-          : Math.max(8_000, Math.ceil((chapterTargetChars + 500) * 2))
-        const targetReserveTokens = Math.ceil(
-          (chapterTargetChars * WRITING_OUTPUT_RESERVE_MULTIPLIER) / 1.7,
-        )
-        const outputReserveChars = Math.max(targetReserveTokens, maxOutputTokens) * 4
-        const scaffold = Math.max(8_000, Math.floor(maxCtx * 0.08))
-        expect(packTokens * 4 + outputReserveChars + scaffold).toBeLessThanOrEqual(maxCtx)
+        expect(
+          plan.contextTokenBudget + plan.outputTokens + plan.scaffoldReserveTokens,
+        ).toBeLessThanOrEqual(plan.windowTokens)
       }
       }
     }
     }
   })
   })
@@ -77,7 +70,8 @@ describe("context pack budget contracts", () => {
 
 
   it("deep chapter no longer hard-codes a 32000 context budget", () => {
   it("deep chapter no longer hard-codes a 32000 context budget", () => {
     const source = readFileSync(resolve(__dirname, "./novel/deep-chapter-generation.ts"), "utf8")
     const source = readFileSync(resolve(__dirname, "./novel/deep-chapter-generation.ts"), "utf8")
-    expect(source).toContain("computeWritingContextPackTokenBudget({")
+    expect(source).toContain("planChapterRequestBudget({")
+    expect(source).toContain("max_tokens: chapterGenerationBudget.outputTokens")
     expect(source).not.toContain("DEEP_CHAPTER_CONTEXT_TOKEN_BUDGET")
     expect(source).not.toContain("DEEP_CHAPTER_CONTEXT_TOKEN_BUDGET")
   })
   })
 })
 })

+ 3 - 3
src/lib/context-budget.spec.ts

@@ -19,9 +19,9 @@ describe("computeOutlineIngestBodyBudget", () => {
     expect(small).toBeGreaterThan(0)
     expect(small).toBeGreaterThan(0)
   })
   })
 
 
-  it("applies CJK language scale", () => {
-    const english = computeOutlineIngestBodyBudget(128_000, promptOverhead, 1)
-    const cjk = computeOutlineIngestBodyBudget(128_000, promptOverhead, 0.425)
+  it("gives CJK fewer characters because each token holds less text", () => {
+    const english = computeOutlineIngestBodyBudget(128_000, promptOverhead, 4)
+    const cjk = computeOutlineIngestBodyBudget(128_000, promptOverhead, 1)
     expect(cjk).toBeLessThan(english)
     expect(cjk).toBeLessThan(english)
   })
   })
 })
 })

+ 216 - 73
src/lib/context-budget.test.ts

@@ -1,99 +1,252 @@
 import { describe, it, expect } from "vitest"
 import { describe, it, expect } from "vitest"
 import {
 import {
+  charsPerTokenForLanguage,
   computeContextBudget,
   computeContextBudget,
   computeNovelContextTokenBudget,
   computeNovelContextTokenBudget,
   computeWritingContextPackTokenBudget,
   computeWritingContextPackTokenBudget,
-  contextScaleForLanguage,
   resolveContextPackTokenBudget,
   resolveContextPackTokenBudget,
-  WRITING_OUTPUT_RESERVE_MULTIPLIER,
+  planChapterRequestBudget,
+  planLlmRequestBudget,
+  planOutlineRequestBudget,
+  LlmContextBudgetError,
 } from "./context-budget"
 } from "./context-budget"
 
 
-// The base-math tests pin langScale=1 so they stay deterministic
+// The base-math tests pin charsPerToken explicitly so they stay deterministic
 // regardless of the active UI language (the app defaults to zh).
 // regardless of the active UI language (the app defaults to zh).
 describe("computeContextBudget", () => {
 describe("computeContextBudget", () => {
-  it("falls back to the 200K-char default for falsy input", () => {
+  it("falls back to the 200K-token default for falsy input", () => {
     expect(computeContextBudget(undefined, 1).maxCtx).toBe(204_800)
     expect(computeContextBudget(undefined, 1).maxCtx).toBe(204_800)
     expect(computeContextBudget(0, 1).maxCtx).toBe(204_800)
     expect(computeContextBudget(0, 1).maxCtx).toBe(204_800)
     expect(computeContextBudget(Number.NaN, 1).maxCtx).toBe(204_800)
     expect(computeContextBudget(Number.NaN, 1).maxCtx).toBe(204_800)
   })
   })
 
 
-  it("allocates fractional sub-budgets from the window", () => {
-    const b = computeContextBudget(200_000, 1)
-    expect(b.responseReserve).toBe(30_000)
-    expect(b.indexBudget).toBe(10_000)
-    expect(b.pageBudget).toBe(100_000)
+  it("converts the token window into a character capacity", () => {
+    const b = computeContextBudget(200_000, 4)
+    expect(b.maxCtx).toBe(800_000)
+    expect(b.responseReserve).toBe(120_000)
   })
   })
 })
 })
 
 
-describe("contextScaleForLanguage", () => {
-  it("keeps scale 1 for English and other non-CJK languages", () => {
-    expect(contextScaleForLanguage("en")).toBe(1)
-    expect(contextScaleForLanguage("en-US")).toBe(1)
-    expect(contextScaleForLanguage("fr")).toBe(1)
+describe("shared LLM request budget", () => {
+  it("keeps the token conservation invariant", () => {
+    const plan = planLlmRequestBudget({
+      maxContextSize: 204_800,
+      desiredOutputTokens: 16_384,
+      requestedContextTokens: 40_000,
+      scaffoldReserveTokens: 8_192,
+      minimumContextTokens: 4_000,
+    })
+    expect(plan).toMatchObject({
+      windowTokens: 184_320,
+      outputTokens: 16_384,
+      contextTokenBudget: 40_000,
+      scaffoldReserveTokens: 8_192,
+      inputTokenBudget: 167_936,
+    })
+    expect(
+      plan.outputTokens + plan.contextTokenBudget + plan.scaffoldReserveTokens,
+    ).toBeLessThanOrEqual(plan.windowTokens)
+  })
+
+  it("treats maxContextSize as tokens, not characters", () => {
+    // The window is a token count already; planning must not divide it down.
+    // Before this was fixed a 200K window planned against 51200 tokens, so a
+    // Chinese session could only use a quarter of the model's real capacity.
+    const plan = planLlmRequestBudget({
+      maxContextSize: 204_800,
+      desiredOutputTokens: 8_192,
+      scaffoldReserveTokens: 0,
+    })
+    expect(plan.windowTokens).toBeGreaterThan(180_000)
+    expect(plan.inputTokenBudget).toBeGreaterThan(170_000)
+  })
+
+  it("reduces output but never below 512 before rejecting an impossible window", () => {
+    const reduced = planLlmRequestBudget({
+      maxContextSize: 1_024,
+      desiredOutputTokens: 8_192,
+      scaffoldReserveTokens: 256,
+      minimumContextTokens: 400,
+    })
+    expect(reduced.outputTokens).toBe(512)
+    expect(reduced.contextTokenBudget).toBe(153)
+    expect(() => planLlmRequestBudget({
+      maxContextSize: 512,
+      desiredOutputTokens: 8_192,
+      scaffoldReserveTokens: 64,
+    })).toThrow(LlmContextBudgetError)
+  })
+
+  it("converges output down to the declared output cap", () => {
+    const plan = planLlmRequestBudget({
+      maxContextSize: 1_000_000,
+      desiredOutputTokens: 150_000,
+      scaffoldReserveTokens: 8_192,
+      maxOutputTokensCap: 65_536,
+    })
+    expect(plan.outputTokens).toBe(65_536)
+  })
+
+  it("raises output to the thinking floor without passing the cap", () => {
+    const raised = planLlmRequestBudget({
+      maxContextSize: 204_800,
+      desiredOutputTokens: 4_000,
+      scaffoldReserveTokens: 8_192,
+      thinkingFloorTokens: 16_384,
+    })
+    expect(raised.outputTokens).toBe(16_384)
+
+    const capped = planLlmRequestBudget({
+      maxContextSize: 204_800,
+      desiredOutputTokens: 4_000,
+      scaffoldReserveTokens: 8_192,
+      thinkingFloorTokens: 16_384,
+      maxOutputTokensCap: 8_192,
+    })
+    expect(capped.outputTokens).toBe(8_192)
+  })
+})
+
+describe("outline request budget", () => {
+  it("scales the output with the window instead of stepping through tiers", () => {
+    expect(planOutlineRequestBudget({
+      maxContextSize: 204_800,
+      stage: "analysis",
+    }).outputTokens).toBe(8_192)
+    expect(planOutlineRequestBudget({
+      maxContextSize: 204_800,
+      stage: "generation",
+    }).outputTokens).toBe(30_720)
+    expect(planOutlineRequestBudget({
+      maxContextSize: 1_000_000,
+      stage: "generation",
+    }).outputTokens).toBe(150_000)
+  })
+
+  it("bounds the generation output by the declared output cap", () => {
+    expect(planOutlineRequestBudget({
+      maxContextSize: 1_000_000,
+      stage: "generation",
+      maxOutputTokens: 65_536,
+    }).outputTokens).toBe(65_536)
+  })
+
+  it("silently raises a legacy 128K window to 204800", () => {
+    const plan = planOutlineRequestBudget({
+      maxContextSize: 128_000,
+      stage: "generation",
+    })
+    expect(plan.windowTokens).toBe(184_320)
+    expect(plan.outputTokens).toBe(30_720)
+  })
+})
+
+describe("chapter request budget", () => {
+  it.each([
+    [2_000, 8_000],
+    [3_000, 8_000],
+    [6_000, 13_000],
+  ])("derives generation output for %i target chars", (chapterTargetChars, outputTokens) => {
+    const plan = planChapterRequestBudget({
+      maxContextSize: 204_800,
+      chapterTargetChars,
+      stage: "generation",
+    })
+    expect(plan.outputTokens).toBe(outputTokens)
+    expect(
+      plan.outputTokens + plan.contextTokenBudget + plan.scaffoldReserveTokens,
+    ).toBeLessThanOrEqual(plan.windowTokens)
+  })
+
+  it("keeps chapter output tied to target length, not to the window", () => {
+    // A 3000-character chapter needs the same output on a 1M model as on a 200K
+    // one, so the generous window must not inflate the request.
+    expect(planChapterRequestBudget({
+      maxContextSize: 1_000_000,
+      chapterTargetChars: 3_000,
+      stage: "generation",
+    }).outputTokens).toBe(8_000)
+  })
+
+  it("bounds the generation output by the declared output cap", () => {
+    expect(planChapterRequestBudget({
+      maxContextSize: 204_800,
+      chapterTargetChars: 6_000,
+      stage: "generation",
+      maxOutputTokens: 4_096,
+    }).outputTokens).toBe(4_096)
+  })
+
+  it("uses 4096 tokens for task analysis", () => {
+    expect(planChapterRequestBudget({
+      maxContextSize: 204_800,
+      chapterTargetChars: 3_000,
+      stage: "analysis",
+    }).outputTokens).toBe(4_096)
+  })
+})
+
+describe("charsPerTokenForLanguage", () => {
+  it("uses the 4:1 ratio for English and other non-CJK languages", () => {
+    expect(charsPerTokenForLanguage("en")).toBe(4)
+    expect(charsPerTokenForLanguage("en-US")).toBe(4)
+    expect(charsPerTokenForLanguage("fr")).toBe(4)
   })
   })
 
 
   it("falls back to the active UI language when none is given", () => {
   it("falls back to the active UI language when none is given", () => {
-    // Test env initialises i18n to zh, so the implicit lookup is CJK-scaled.
-    expect(contextScaleForLanguage()).toBeCloseTo(0.425, 5)
+    // Test env initialises i18n to zh, so the implicit lookup is CJK.
+    expect(charsPerTokenForLanguage()).toBe(1)
   })
   })
 
 
-  it("shrinks the window for CJK languages", () => {
-    expect(contextScaleForLanguage("zh")).toBeCloseTo(0.425, 5)
-    expect(contextScaleForLanguage("zh-CN")).toBeCloseTo(0.425, 5)
-    expect(contextScaleForLanguage("ja")).toBeCloseTo(0.425, 5)
-    expect(contextScaleForLanguage("ko")).toBeCloseTo(0.425, 5)
+  it("counts one character per token for CJK languages", () => {
+    // Must match the token estimator, which also counts 1 CJK char = 1 token.
+    expect(charsPerTokenForLanguage("zh")).toBe(1)
+    expect(charsPerTokenForLanguage("zh-CN")).toBe(1)
+    expect(charsPerTokenForLanguage("ja")).toBe(1)
+    expect(charsPerTokenForLanguage("ko")).toBe(1)
   })
   })
 })
 })
 
 
 describe("computeContextBudget language scaling", () => {
 describe("computeContextBudget language scaling", () => {
-  it("scales the effective window down for CJK UIs", () => {
-    const zh = contextScaleForLanguage("zh")
-    expect(computeContextBudget(200_000, zh).maxCtx).toBe(85_000)
-    expect(computeContextBudget(204_800, zh).maxCtx).toBe(87_040)
+  it("yields fewer characters for CJK because each token holds less", () => {
+    const zh = charsPerTokenForLanguage("zh")
+    expect(computeContextBudget(200_000, zh).maxCtx).toBe(200_000)
+    expect(computeContextBudget(204_800, zh).maxCtx).toBe(204_800)
   })
   })
 
 
-  it("leaves English windows untouched", () => {
-    expect(computeContextBudget(200_000, contextScaleForLanguage("en")).maxCtx).toBe(200_000)
+  it("yields four characters per token for English", () => {
+    expect(computeContextBudget(200_000, charsPerTokenForLanguage("en")).maxCtx).toBe(800_000)
   })
   })
 })
 })
 
 
 describe("computeNovelContextTokenBudget", () => {
 describe("computeNovelContextTokenBudget", () => {
-  it("preserves the legacy 32K-token deep-chapter budget on the default window", () => {
-    // Default window (204800 chars) → cap 33280 tokens, so 32000 is kept intact.
-    expect(computeNovelContextTokenBudget(204_800, 32_000, 1)).toBe(32_000)
-    expect(computeNovelContextTokenBudget(undefined, 32_000, 1)).toBe(32_000)
+  it("keeps a requested budget that fits under the window share", () => {
+    expect(computeNovelContextTokenBudget(204_800, 32_000)).toBe(32_000)
+    expect(computeNovelContextTokenBudget(undefined, 32_000)).toBe(32_000)
   })
   })
 
 
   it("caps an unset (0 / unlimited) budget at the window-derived ceiling", () => {
   it("caps an unset (0 / unlimited) budget at the window-derived ceiling", () => {
-    expect(computeNovelContextTokenBudget(204_800, 0, 1)).toBe(33_280)
-    expect(computeNovelContextTokenBudget(204_800, undefined, 1)).toBe(33_280)
+    expect(computeNovelContextTokenBudget(204_800, 0)).toBe(133_120)
+    expect(computeNovelContextTokenBudget(204_800, undefined)).toBe(133_120)
   })
   })
 
 
   it("clamps an over-large user budget down to the ceiling", () => {
   it("clamps an over-large user budget down to the ceiling", () => {
-    expect(computeNovelContextTokenBudget(204_800, 100_000, 1)).toBe(33_280)
+    expect(computeNovelContextTokenBudget(204_800, 200_000)).toBe(133_120)
   })
   })
 
 
   it("shrinks the budget proportionally for small windows", () => {
   it("shrinks the budget proportionally for small windows", () => {
-    // 32000 chars → floor(32000 * 0.65 / 4) = 5200 tokens.
-    expect(computeNovelContextTokenBudget(32_000, 32_000, 1)).toBe(5_200)
+    expect(computeNovelContextTokenBudget(32_000, 32_000)).toBe(20_800)
   })
   })
 
 
   it("never drops below the token floor", () => {
   it("never drops below the token floor", () => {
-    expect(computeNovelContextTokenBudget(1_000, 0, 1)).toBe(4_000)
-  })
-
-  it("tightens the ceiling for CJK UIs so the same request is capped down", () => {
-    // zh: maxCtx 204800*0.425=87040 → cap floor(87040*0.65/4)=14144 tokens.
-    const zh = contextScaleForLanguage("zh")
-    expect(computeNovelContextTokenBudget(204_800, 32_000, zh)).toBe(14_144)
-    expect(computeNovelContextTokenBudget(204_800, 0, zh)).toBe(14_144)
+    expect(computeNovelContextTokenBudget(1_000, 0)).toBe(4_000)
   })
   })
 })
 })
 
 
 describe("resolveContextPackTokenBudget", () => {
 describe("resolveContextPackTokenBudget", () => {
   it("always returns a positive finite budget for auto mode", () => {
   it("always returns a positive finite budget for auto mode", () => {
-    const budget = resolveContextPackTokenBudget({ maxContextSize: 204_800, contextTokenBudget: 0, langScale: 1 })
-    expect(budget).toBe(33_280)
+    const budget = resolveContextPackTokenBudget({ maxContextSize: 204_800, contextTokenBudget: 0 })
+    expect(budget).toBe(133_120)
     expect(Number.isFinite(budget)).toBe(true)
     expect(Number.isFinite(budget)).toBe(true)
   })
   })
 
 
@@ -101,79 +254,69 @@ describe("resolveContextPackTokenBudget", () => {
     expect(resolveContextPackTokenBudget({
     expect(resolveContextPackTokenBudget({
       maxContextSize: 204_800,
       maxContextSize: 204_800,
       contextTokenBudget: 10_000,
       contextTokenBudget: 10_000,
-      langScale: 1,
     })).toBe(10_000)
     })).toBe(10_000)
   })
   })
 })
 })
 
 
 describe("computeWritingContextPackTokenBudget", () => {
 describe("computeWritingContextPackTokenBudget", () => {
-  it("reserves at least maxOutputTokens (as chars) before allocating the pack", () => {
-    const maxContextSize = 204_800
-    const chapterTargetChars = 3_000
-    const langScale = 1
-    const { maxCtx } = computeContextBudget(maxContextSize, langScale)
+  it("leaves room for the chapter output and scaffolding inside the window", () => {
+    const plan = planChapterRequestBudget({
+      maxContextSize: 204_800,
+      contextTokenBudget: 0,
+      chapterTargetChars: 3_000,
+      stage: "generation",
+    })
     const budget = computeWritingContextPackTokenBudget({
     const budget = computeWritingContextPackTokenBudget({
-      maxContextSize,
+      maxContextSize: 204_800,
       contextTokenBudget: 0,
       contextTokenBudget: 0,
-      chapterTargetChars,
-      langScale,
+      chapterTargetChars: 3_000,
     })
     })
-    const maxOutputTokens = 8_000
-    const targetReserveTokens = Math.ceil((chapterTargetChars * WRITING_OUTPUT_RESERVE_MULTIPLIER) / 1.7)
-    const outputReserveChars = Math.max(targetReserveTokens, maxOutputTokens) * 4
-    const scaffold = Math.max(8_000, Math.floor(maxCtx * 0.08))
-    expect(budget * 4 + outputReserveChars + scaffold).toBeLessThanOrEqual(maxCtx)
-    expect(outputReserveChars).toBeGreaterThanOrEqual(maxOutputTokens * 4)
+    expect(budget).toBe(plan.contextTokenBudget)
+    expect(budget + plan.outputTokens + plan.scaffoldReserveTokens)
+      .toBeLessThanOrEqual(plan.windowTokens)
   })
   })
 
 
   it("shrinks when chapter target grows", () => {
   it("shrinks when chapter target grows", () => {
     const smallTarget = computeWritingContextPackTokenBudget({
     const smallTarget = computeWritingContextPackTokenBudget({
       maxContextSize: 204_800,
       maxContextSize: 204_800,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })
     })
     const largeTarget = computeWritingContextPackTokenBudget({
     const largeTarget = computeWritingContextPackTokenBudget({
       maxContextSize: 204_800,
       maxContextSize: 204_800,
       chapterTargetChars: 6_000,
       chapterTargetChars: 6_000,
-      langScale: 1,
     })
     })
     expect(largeTarget).toBeLessThanOrEqual(smallTarget)
     expect(largeTarget).toBeLessThanOrEqual(smallTarget)
   })
   })
 
 
   it("grows with a larger window within the general cap", () => {
   it("grows with a larger window within the general cap", () => {
     const smallWindow = computeWritingContextPackTokenBudget({
     const smallWindow = computeWritingContextPackTokenBudget({
-      maxContextSize: 64_000,
+      maxContextSize: 204_800,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })
     })
     const largeWindow = computeWritingContextPackTokenBudget({
     const largeWindow = computeWritingContextPackTokenBudget({
-      maxContextSize: 204_800,
+      maxContextSize: 1_000_000,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })
     })
-    expect(largeWindow).toBeGreaterThanOrEqual(smallWindow)
+    expect(largeWindow).toBeGreaterThan(smallWindow)
   })
   })
 
 
   it("never exceeds the general window cap", () => {
   it("never exceeds the general window cap", () => {
     const budget = computeWritingContextPackTokenBudget({
     const budget = computeWritingContextPackTokenBudget({
       maxContextSize: 204_800,
       maxContextSize: 204_800,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })
     })
-    expect(budget).toBeLessThanOrEqual(computeNovelContextTokenBudget(204_800, 0, 1))
+    expect(budget).toBeLessThanOrEqual(computeNovelContextTokenBudget(204_800, 0))
   })
   })
 
 
   it("clamps an explicit user budget to the writing-derived auto budget", () => {
   it("clamps an explicit user budget to the writing-derived auto budget", () => {
     const auto = computeWritingContextPackTokenBudget({
     const auto = computeWritingContextPackTokenBudget({
       maxContextSize: 204_800,
       maxContextSize: 204_800,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })
     })
     expect(computeWritingContextPackTokenBudget({
     expect(computeWritingContextPackTokenBudget({
       maxContextSize: 204_800,
       maxContextSize: 204_800,
-      contextTokenBudget: 100_000,
+      contextTokenBudget: 300_000,
       chapterTargetChars: 3_000,
       chapterTargetChars: 3_000,
-      langScale: 1,
     })).toBe(auto)
     })).toBe(auto)
   })
   })
 })
 })

+ 242 - 140
src/lib/context-budget.ts

@@ -1,77 +1,50 @@
 /**
 /**
- * Pure budget allocator for chat context assembly.
+ * Pure budget allocator for LLM request assembly.
  *
  *
- * Given an LLM's `maxContextSize` (in characters — see wiki-store.ts;
- * yes, that's a quirky unit, but tokens-vs-chars conversion lives
- * elsewhere), compute the per-section character budgets used by
- * chat-panel when packing the prompt.
+ * `maxContextSize` is the model's context window in TOKENS — it is copied
+ * straight from the provider's spec sheet (Gemini 1M, Kimi 256K, …) by the
+ * settings UI. Two domains are derived from it here:
  *
  *
- * Why this is its own module:
- *   - The math has corner cases that deserve their own tests
- *     (tiny configs, huge configs, the legacy 30K cap removal).
- *   - Inlining it in chat-panel.tsx made it untestable in isolation.
+ *   - Token domain (`planLlmRequestBudget` and the chapter/outline planners):
+ *     works in the same unit as the window, so no conversion happens at all.
+ *     This is the authoritative allocator — it guarantees input + output fit.
+ *   - Character domain (`computeContextBudget`): converts the token window
+ *     into how many CHARACTERS of prompt text will fit, for the callers that
+ *     slice raw strings. The conversion rate is language-dependent, which is
+ *     what `charsPerTokenForLanguage` supplies.
  *
  *
- * The shape of the budget:
- *
- *   ┌─────────────────────────────────────────────────────┐
- *   │              maxCtx (100%)                          │
- *   ├──────┬───────────────┬──────────────────┬───────────┤
- *   │ idx  │   pages       │  history + sys   │  resp     │
- *   │  5%  │    50%        │    ~30%          │   15%     │
- *   └──────┴───────────────┴──────────────────┴───────────┘
- *
- * `historyAndSystem` isn't returned because it's not enforced as a
- * single budget — system prompt is roughly fixed-size, and history
- * is gated by `maxHistoryMessages` (count, not bytes). The leftover
- * just provides headroom.
- *
- * The response reserve is a "passive" reservation: we don't pass
- * `max_tokens: responseReserve / 3` to the LLM (yet — that's a
- * follow-up). We just refuse to fill above (maxCtx - responseReserve)
- * so the LLM has room to actually answer.
+ * The two must never both apply a language factor to the same value: the
+ * token domain already speaks tokens, so scaling it by language would count
+ * the same density twice.
  */
  */
 
 
 import i18n from "@/i18n"
 import i18n from "@/i18n"
+import { normalizeUserLlmContextSize } from "@/lib/llm-context-size"
 
 
 /** Result of `computeContextBudget`. All values are character counts. */
 /** Result of `computeContextBudget`. All values are character counts. */
 export interface ContextBudget {
 export interface ContextBudget {
-  /** The model's full context window (always populated; falls back
-   *  to a sensible default when caller passes 0/undefined). */
+  /** How many characters of prompt text the model's token window holds,
+   *  at the active language's density. Falls back to a sensible default
+   *  when the caller passes 0/undefined. */
   maxCtx: number
   maxCtx: number
   /** Characters NOT to be filled with prompt content — left empty so
   /** Characters NOT to be filled with prompt content — left empty so
    *  the LLM has room to write its response. */
    *  the LLM has room to write its response. */
   responseReserve: number
   responseReserve: number
-  /** Wiki index summary budget. ~5% — enough to list every page's
-   *  title without occupying serious budget. */
-  indexBudget: number
-  /** Total characters available for retrieved wiki page content. */
-  pageBudget: number
-  /** Per-page truncation cap. A single page won't be embedded longer
-   *  than this even if `pageBudget` would allow it. Scales with
-   *  pageBudget (used to be hard-capped at 30,000 chars regardless
-   *  of context size — that wasted budget on long-context models). */
-  maxPageSize: number
 }
 }
 
 
 const DEFAULT_MAX_CTX = 204_800
 const DEFAULT_MAX_CTX = 204_800
-const RESPONSE_RESERVE_FRAC = 0.15
-const INDEX_BUDGET_FRAC = 0.05
-const PAGE_BUDGET_FRAC = 0.5
-const PER_PAGE_FRAC = 0.3
-const PER_PAGE_FLOOR = 5_000
-
-/** Approximate characters per token the whole budgeting layer assumes.
- *  `maxContextSize` is expressed in CHARACTERS under the English-ish
- *  assumption of ~4 chars/token (see contextPackToPrompt). */
+export const RESPONSE_RESERVE_FRAC = 0.15
+
+/** Characters per token for English-ish text — the conventional 4:1. */
 const CHARS_PER_TOKEN = 4
 const CHARS_PER_TOKEN = 4
-/** Empirical chars/token for CJK (Chinese/Japanese/Korean) text. CJK is
- *  ~2.3x denser than English, so the same character budget maps to far
- *  more tokens and can overflow the model window. */
-const CHARS_PER_TOKEN_CJK = 1.7
-/** Effective-window multiplier for CJK UIs. Shrinks the character budget
- *  so its TOKEN footprint matches what the English assumption expects,
- *  keeping token usage comparable across languages. ≈ 0.425. */
-const CJK_CONTEXT_SCALE = CHARS_PER_TOKEN_CJK / CHARS_PER_TOKEN
+/** Characters per token for CJK text. Deliberately 1.0 to match
+ *  `src/lib/context-hub/token-estimator.ts`, which counts one CJK character
+ *  as one token. A looser value here would let the character budgets admit
+ *  more text than the token estimator allows, so the surplus would be packed
+ *  in and then trimmed back out in `streamChat` — wasted work and lost
+ *  content. Real tokenizers land around 1–1.5 chars/token, so 1.0 is the
+ *  safe end. */
+const CHARS_PER_TOKEN_CJK = 1
 
 
 function isCjkLanguage(lang: string | undefined): boolean {
 function isCjkLanguage(lang: string | undefined): boolean {
   if (!lang) return false
   if (!lang) return false
@@ -80,59 +53,39 @@ function isCjkLanguage(lang: string | undefined): boolean {
 }
 }
 
 
 /**
 /**
- * Window scale for a UI language. English (and any non-CJK language)
- * returns 1 → zero behavioural change. CJK returns `CJK_CONTEXT_SCALE`
- * so the character budgets translate to a safe token footprint.
+ * How many characters one token holds in a given UI language, used to turn
+ * the model's token window into a character budget.
  *
  *
  * `lang` defaults to the active i18n language; pass an explicit value
  * `lang` defaults to the active i18n language; pass an explicit value
  * (e.g. in tests) to keep the calculation deterministic.
  * (e.g. in tests) to keep the calculation deterministic.
  */
  */
-export function contextScaleForLanguage(lang?: string): number {
+export function charsPerTokenForLanguage(lang?: string): number {
   const resolved =
   const resolved =
     lang ?? (typeof i18n?.language === "string" ? i18n.language : undefined)
     lang ?? (typeof i18n?.language === "string" ? i18n.language : undefined)
-  return isCjkLanguage(resolved) ? CJK_CONTEXT_SCALE : 1
+  return isCjkLanguage(resolved) ? CHARS_PER_TOKEN_CJK : CHARS_PER_TOKEN
 }
 }
 
 
 /**
 /**
- * Compute character budgets from the LLM's max context window.
+ * Convert the model's token window into character budgets.
  *
  *
- * Falsy `maxContextSize` (0 / NaN / undefined) falls back to the
- * pre-Phase-1 default of 200K chars so existing configs don't break.
+ * Falsy `maxContextSize` (0 / NaN / undefined) falls back to the default
+ * 200K-token window so existing configs don't break.
  */
  */
 export function computeContextBudget(
 export function computeContextBudget(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
-  langScale: number = contextScaleForLanguage(),
+  charsPerToken: number = charsPerTokenForLanguage(),
 ): ContextBudget {
 ): ContextBudget {
-  const rawMaxCtx =
+  const windowTokens =
     typeof maxContextSize === "number" && maxContextSize > 0
     typeof maxContextSize === "number" && maxContextSize > 0
       ? maxContextSize
       ? maxContextSize
       : DEFAULT_MAX_CTX
       : DEFAULT_MAX_CTX
-  const scale = typeof langScale === "number" && langScale > 0 ? langScale : 1
-  const maxCtx = Math.max(1, Math.floor(rawMaxCtx * scale))
-
-  const responseReserve = Math.floor(maxCtx * RESPONSE_RESERVE_FRAC)
-  const indexBudget = Math.floor(maxCtx * INDEX_BUDGET_FRAC)
-  const pageBudget = Math.floor(maxCtx * PAGE_BUDGET_FRAC)
-
-  // Per-page cap rules:
-  //   - At minimum, allow PER_PAGE_FLOOR (5K) so a small config still
-  //     fits one short page.
-  //   - At maximum, never exceed pageBudget itself — for tiny configs
-  //     where pageBudget < 5K, the floor would otherwise allow a
-  //     single page bigger than the entire page budget, which then
-  //     gets entirely rejected by tryAddPage in chat-panel.
-  //   - Otherwise scale linearly with pageBudget at PER_PAGE_FRAC (30%).
-  const maxPageSize = Math.min(
-    pageBudget,
-    Math.max(PER_PAGE_FLOOR, Math.floor(pageBudget * PER_PAGE_FRAC)),
-  )
+  const density =
+    typeof charsPerToken === "number" && charsPerToken > 0 ? charsPerToken : CHARS_PER_TOKEN
+  const maxCtx = Math.max(1, Math.floor(windowTokens * density))
 
 
   return {
   return {
     maxCtx,
     maxCtx,
-    responseReserve,
-    indexBudget,
-    pageBudget,
-    maxPageSize,
+    responseReserve: Math.floor(maxCtx * RESPONSE_RESERVE_FRAC),
   }
   }
 }
 }
 
 
@@ -156,18 +109,21 @@ const NOVEL_CONTEXT_TOKEN_FLOOR = 4_000
  * clamped to the window-derived cap; when unset the cap itself is used so
  * clamped to the window-derived cap; when unset the cap itself is used so
  * the injection is never truly unbounded.
  * the injection is never truly unbounded.
  *
  *
- * Unit note: `maxContextSize` is in CHARACTERS while `contextPackToPrompt`
- * expects a TOKEN budget (~4 chars/token), hence the division.
+ * Stays entirely in the token domain: the window is already tokens and the
+ * consumer wants tokens, so there is no character round-trip and no language
+ * factor. Language density is the token estimator's job.
  */
  */
 export function computeNovelContextTokenBudget(
 export function computeNovelContextTokenBudget(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
   requestedTokenBudget?: number,
   requestedTokenBudget?: number,
-  langScale?: number,
 ): number {
 ): number {
-  const { maxCtx } = computeContextBudget(maxContextSize, langScale)
+  const windowTokens =
+    typeof maxContextSize === "number" && maxContextSize > 0
+      ? maxContextSize
+      : DEFAULT_MAX_CTX
   const cap = Math.max(
   const cap = Math.max(
     NOVEL_CONTEXT_TOKEN_FLOOR,
     NOVEL_CONTEXT_TOKEN_FLOOR,
-    Math.floor((maxCtx * NOVEL_CONTEXT_FRAC) / CHARS_PER_TOKEN),
+    Math.floor(windowTokens * NOVEL_CONTEXT_FRAC),
   )
   )
   if (requestedTokenBudget && requestedTokenBudget > 0) {
   if (requestedTokenBudget && requestedTokenBudget > 0) {
     return Math.min(requestedTokenBudget, cap)
     return Math.min(requestedTokenBudget, cap)
@@ -179,7 +135,6 @@ export interface ResolveContextPackTokenBudgetInput {
   maxContextSize?: number
   maxContextSize?: number
   /** User setting; 0 / undefined = auto from window. */
   /** User setting; 0 / undefined = auto from window. */
   contextTokenBudget?: number
   contextTokenBudget?: number
-  langScale?: number
 }
 }
 
 
 /**
 /**
@@ -192,63 +147,210 @@ export function resolveContextPackTokenBudget(
   return computeNovelContextTokenBudget(
   return computeNovelContextTokenBudget(
     input.maxContextSize,
     input.maxContextSize,
     input.contextTokenBudget,
     input.contextTokenBudget,
-    input.langScale,
   )
   )
 }
 }
 
 
-/** Output reserve multiplier: chapter target chars × this factor. */
+export const MIN_LLM_OUTPUT_TOKENS = 512
+
+export class LlmContextBudgetError extends Error {
+  constructor(message = "模型上下文不足:无法同时容纳系统提示、当前用户请求和最小输出空间。") {
+    super(message)
+    this.name = "LlmContextBudgetError"
+  }
+}
+
+/**
+ * Headroom kept between our token estimates and the model's real window.
+ * Estimation is approximate in both directions (tokenizer differences,
+ * scaffolding we don't see), so we plan against 90% of the advertised
+ * window. This replaces an earlier `/ 4`, which looked like a safety factor
+ * but was actually a character-to-token conversion applied to a value that
+ * was already in tokens — shrinking every window to a quarter of its size.
+ */
+const LLM_WINDOW_SAFETY_FRAC = 0.9
+
+export interface LlmRequestBudgetInput {
+  maxContextSize?: number
+  desiredOutputTokens: number
+  requestedContextTokens?: number
+  scaffoldReserveTokens: number
+  minimumContextTokens?: number
+  minimumOutputTokens?: number
+  /** The model's declared maximum output, from the user's settings. Output
+   *  is never planned above this even when the window could hold more. */
+  maxOutputTokensCap?: number
+  /** Output the active reasoning level needs before it can produce any final
+   *  content (`thinkingMinMaxTokens`). Raises the plan, but stays subject to
+   *  the cap and the window — unlike a floor applied to the request body,
+   *  which would silently break the conservation guaranteed here. */
+  thinkingFloorTokens?: number
+}
+
+export interface LlmRequestBudgetPlan {
+  windowTokens: number
+  outputTokens: number
+  contextTokenBudget: number
+  scaffoldReserveTokens: number
+  inputTokenBudget: number
+}
+
+function finiteNonNegative(value: number | undefined, fallback = 0): number {
+  return Number.isFinite(value) && (value as number) > 0
+    ? Math.floor(value as number)
+    : fallback
+}
+
+/** Token-domain conservation kernel shared by chapter and outline workflows. */
+export function planLlmRequestBudget(input: LlmRequestBudgetInput): LlmRequestBudgetPlan {
+  const rawWindow = Number.isFinite(input.maxContextSize) && (input.maxContextSize as number) > 0
+    ? Math.floor(input.maxContextSize as number)
+    : normalizeUserLlmContextSize(undefined)
+  const windowTokens = Math.max(1, Math.floor(rawWindow * LLM_WINDOW_SAFETY_FRAC))
+  const scaffoldReserveTokens = finiteNonNegative(input.scaffoldReserveTokens)
+  const minimumOutputTokens = Math.max(
+    MIN_LLM_OUTPUT_TOKENS,
+    finiteNonNegative(input.minimumOutputTokens, MIN_LLM_OUTPUT_TOKENS),
+  )
+  const outputCap = finiteNonNegative(input.maxOutputTokensCap, Number.MAX_SAFE_INTEGER)
+  const desiredOutputTokens = Math.max(
+    minimumOutputTokens,
+    finiteNonNegative(input.desiredOutputTokens, minimumOutputTokens),
+  )
+  // The thinking floor may not push output past what the model can emit.
+  const thinkingFloorTokens = Math.min(finiteNonNegative(input.thinkingFloorTokens), outputCap)
+  const targetOutputTokens = Math.max(desiredOutputTokens, thinkingFloorTokens)
+  const minimumContextTokens = finiteNonNegative(input.minimumContextTokens)
+  const available = windowTokens - scaffoldReserveTokens
+  if (available < minimumOutputTokens) throw new LlmContextBudgetError()
+
+  // Keep the requested minimum context where possible, then allocate output.
+  // If both cannot fit, context is the degradable side; output never drops below 512.
+  const outputCeiling = Math.max(
+    minimumOutputTokens,
+    Math.min(outputCap, available - minimumContextTokens),
+  )
+  const outputTokens = Math.min(targetOutputTokens, outputCeiling)
+  const remainingForContext = Math.max(0, available - outputTokens)
+  const requestedContextTokens = finiteNonNegative(input.requestedContextTokens)
+  const contextTokenBudget = requestedContextTokens > 0
+    ? Math.min(requestedContextTokens, remainingForContext)
+    : remainingForContext
+  const inputTokenBudget = windowTokens - outputTokens
+
+  return {
+    windowTokens,
+    outputTokens,
+    contextTokenBudget,
+    scaffoldReserveTokens,
+    inputTokenBudget,
+  }
+}
+
+export type ChapterBudgetStage = "analysis" | "generation"
+
+export interface PlanChapterRequestBudgetInput {
+  maxContextSize?: number
+  contextTokenBudget?: number
+  chapterTargetChars?: number
+  stage: ChapterBudgetStage
+  maxOutputTokens?: number
+  thinkingFloorTokens?: number
+}
+
+function chapterMaxOutputTokens(targetChars?: number): number {
+  const target = Number.isFinite(targetChars) && (targetChars as number) > 0
+    ? Math.max(2_000, Math.min(6_000, Math.round(targetChars as number)))
+    : 3_000
+  return target === 3_000 ? 8_000 : Math.max(8_000, Math.ceil((target + 500) * 2))
+}
+
+export function planChapterRequestBudget(
+  input: PlanChapterRequestBudgetInput,
+): LlmRequestBudgetPlan {
+  const normalizedWindow = normalizeUserLlmContextSize(input.maxContextSize)
+  const genericContextCap = computeNovelContextTokenBudget(
+    normalizedWindow,
+    input.contextTokenBudget,
+  )
+  // Chapter output is sized from the user's target chapter length rather than
+  // a share of the window: a 3000-character chapter needs the same output on a
+  // 200K model as on a 1M one.
+  return planLlmRequestBudget({
+    maxContextSize: normalizedWindow,
+    desiredOutputTokens: input.stage === "analysis"
+      ? 4_096
+      : chapterMaxOutputTokens(input.chapterTargetChars),
+    requestedContextTokens: genericContextCap,
+    scaffoldReserveTokens: 8_000,
+    minimumContextTokens: 2_000,
+    maxOutputTokensCap: input.maxOutputTokens,
+    thinkingFloorTokens: input.thinkingFloorTokens,
+  })
+}
+
+export type OutlineBudgetStage = "analysis" | "generation"
+
+/** Share of the window the outline's own response may claim. Reuses the
+ *  response reserve the rest of the budgeting layer already assumes. */
+const OUTLINE_GENERATION_OUTPUT_FRAC = RESPONSE_RESERVE_FRAC
+/** Analysis passes summarise rather than draft, so they need far less. */
+const OUTLINE_ANALYSIS_OUTPUT_FRAC = 0.04
+
+export interface PlanOutlineRequestBudgetInput {
+  maxContextSize?: number
+  contextTokenBudget?: number
+  stage: OutlineBudgetStage
+  maxOutputTokens?: number
+  thinkingFloorTokens?: number
+}
+
+export function planOutlineRequestBudget(
+  input: PlanOutlineRequestBudgetInput,
+): LlmRequestBudgetPlan {
+  const normalizedWindow = normalizeUserLlmContextSize(input.maxContextSize)
+  // Scales with the window instead of stepping through fixed tiers, and is
+  // then bounded by the user's declared output cap inside the kernel.
+  const desiredOutputTokens = Math.floor(normalizedWindow * (input.stage === "analysis"
+    ? OUTLINE_ANALYSIS_OUTPUT_FRAC
+    : OUTLINE_GENERATION_OUTPUT_FRAC))
+  const genericContextCap = computeNovelContextTokenBudget(
+    normalizedWindow,
+    input.contextTokenBudget,
+  )
+  return planLlmRequestBudget({
+    maxContextSize: normalizedWindow,
+    desiredOutputTokens,
+    requestedContextTokens: genericContextCap,
+    scaffoldReserveTokens: 8_192,
+    minimumContextTokens: 4_000,
+    maxOutputTokensCap: input.maxOutputTokens,
+    thinkingFloorTokens: input.thinkingFloorTokens,
+  })
+}
+
+/** Legacy compatibility constant retained for callers/tests that compare old reserves. */
 export const WRITING_OUTPUT_RESERVE_MULTIPLIER = 2
 export const WRITING_OUTPUT_RESERVE_MULTIPLIER = 2
-/** Minimum scaffold reserve for writing prompts (instructions / outline shell). */
-const WRITING_SCAFFOLD_RESERVE_FLOOR = 8_000
-const WRITING_SCAFFOLD_RESERVE_FRAC = 0.08
 
 
 export interface ComputeWritingContextPackTokenBudgetInput {
 export interface ComputeWritingContextPackTokenBudgetInput {
   maxContextSize?: number
   maxContextSize?: number
   contextTokenBudget?: number
   contextTokenBudget?: number
   chapterTargetChars?: number
   chapterTargetChars?: number
-  langScale?: number
+  maxOutputTokens?: number
 }
 }
 
 
 /**
 /**
- * Deep-chapter ContextPack budget: window minus output reserve (target×2)
- * and scaffold, then clamped by the general window cap / user budget.
+ * Compatibility wrapper over the shared chapter-generation budget strategy.
  */
  */
 export function computeWritingContextPackTokenBudget(
 export function computeWritingContextPackTokenBudget(
   input: ComputeWritingContextPackTokenBudgetInput,
   input: ComputeWritingContextPackTokenBudgetInput,
 ): number {
 ): number {
-  const langScale = input.langScale
-  const { maxCtx } = computeContextBudget(input.maxContextSize, langScale)
-  // Inline clamp mirrors resolveChapterLengthSpec without importing deep-chapter-prompts
-  // (avoids circular deps). Keep in sync with DEEP_CHAPTER 2000–6000 bounds.
-  const rawTarget = input.chapterTargetChars
-  const target = Number.isFinite(rawTarget) && (rawTarget as number) > 0
-    ? Math.max(2_000, Math.min(6_000, Math.round(rawTarget as number)))
-    : 3_000
-  // Same formula as resolveChapterLengthSpec.maxOutputTokens.
-  const maxOutputTokens = target === 3_000
-    ? 8_000
-    : Math.max(8_000, Math.ceil((target + 500) * 2))
-  // User redundancy: 2× target chars at CJK density (~1.7 chars/token), then take the
-  // larger of that vs the chapter maxOutputTokens so generation headroom is real.
-  const targetReserveTokens = Math.ceil(
-    (target * WRITING_OUTPUT_RESERVE_MULTIPLIER) / CHARS_PER_TOKEN_CJK,
-  )
-  const outputReserveTokens = Math.max(targetReserveTokens, maxOutputTokens)
-  const outputReserveChars = outputReserveTokens * CHARS_PER_TOKEN
-  const scaffoldReserveChars = Math.max(
-    WRITING_SCAFFOLD_RESERVE_FLOOR,
-    Math.floor(maxCtx * WRITING_SCAFFOLD_RESERVE_FRAC),
-  )
-  const availableChars = Math.max(0, maxCtx - outputReserveChars - scaffoldReserveChars)
-  // Do not inflate with NOVEL_CONTEXT_TOKEN_FLOOR: that would steal the output reserve
-  // on small windows. Prefer leaving room for chapter generation.
-  const derivedTokens = Math.max(0, Math.floor(availableChars / CHARS_PER_TOKEN))
-  const windowCap = computeNovelContextTokenBudget(input.maxContextSize, 0, langScale)
-  const autoTokens = Math.min(derivedTokens, windowCap)
-  if (input.contextTokenBudget && input.contextTokenBudget > 0) {
-    return Math.min(input.contextTokenBudget, autoTokens)
-  }
-  return autoTokens
+  return planChapterRequestBudget({
+    maxContextSize: input.maxContextSize,
+    contextTokenBudget: input.contextTokenBudget,
+    chapterTargetChars: input.chapterTargetChars,
+    stage: "generation",
+    maxOutputTokens: input.maxOutputTokens,
+  }).contextTokenBudget
 }
 }
 
 
 /** Legacy single-pass outline ingest floor; kept so small windows still behave predictably. */
 /** Legacy single-pass outline ingest floor; kept so small windows still behave predictably. */
@@ -264,15 +366,15 @@ function clampBudget(value: number, min: number, max: number): number {
  * Character budget for the outline body in `ingestOutline`.
  * Character budget for the outline body in `ingestOutline`.
  *
  *
  * Reserves space for fixed prompts and JSON output, then allocates the
  * Reserves space for fixed prompts and JSON output, then allocates the
- * remainder to the outline markdown. Scales with `maxContextSize` and
- * CJK language scale like other budget helpers.
+ * remainder to the outline markdown. Scales with `maxContextSize` and the
+ * active language's character density like other character-domain helpers.
  */
  */
 export function computeOutlineIngestBodyBudget(
 export function computeOutlineIngestBodyBudget(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
   promptOverheadChars: number,
   promptOverheadChars: number,
-  langScale?: number,
+  charsPerToken?: number,
 ): number {
 ): number {
-  const { maxCtx, responseReserve } = computeContextBudget(maxContextSize, langScale)
+  const { maxCtx, responseReserve } = computeContextBudget(maxContextSize, charsPerToken)
   const outputReserve = Math.max(responseReserve, Math.floor(maxCtx * 0.15))
   const outputReserve = Math.max(responseReserve, Math.floor(maxCtx * 0.15))
   const instructionReserve = Math.max(promptOverheadChars, Math.floor(maxCtx * 0.08))
   const instructionReserve = Math.max(promptOverheadChars, Math.floor(maxCtx * 0.08))
   const available = maxCtx - outputReserve - instructionReserve
   const available = maxCtx - outputReserve - instructionReserve

+ 6 - 3
src/lib/context-hub/ai-outline-integration.spec.ts

@@ -34,9 +34,12 @@ describe("AI outline context hub integration", () => {
     expect(source).toContain("contextHub.saveSnapshot(`${messageId}:${runId}`, contextHubResult)")
     expect(source).toContain("contextHub.saveSnapshot(`${messageId}:${runId}`, contextHubResult)")
   })
   })
 
 
-  it("passes model window size so unlimited token budget scales safely", () => {
-    expect(source).toContain("tokenBudget: novelConfig.contextTokenBudget,")
+  it("passes the shared workflow budget and model window to Context Hub", () => {
+    expect(source).toContain("tokenBudget: outlineRequestBudget.contextTokenBudget,")
+    expect(source).toContain("tokenBudget: resumeRequestBudget.contextTokenBudget,")
+    expect(source).toContain("tokenBudget: regenerationRequestBudget.contextTokenBudget,")
     expect(source).toContain("maxContextSize: effectiveLlmConfig.maxContextSize,")
     expect(source).toContain("maxContextSize: effectiveLlmConfig.maxContextSize,")
-    expect(source).not.toContain("contextTokenBudget > 0")
+    expect(source).toContain("max_tokens: requestBudget.outputTokens")
+    expect(source).not.toContain("max_tokens: Math.max(")
   })
   })
 })
 })

+ 1 - 1
src/lib/context-hub/composer.ts

@@ -11,7 +11,7 @@ export interface ComposeContextInput {
   confidence?: number
   confidence?: number
   /** Explicit token budget; 0 / undefined = window-derived safe cap. */
   /** Explicit token budget; 0 / undefined = window-derived safe cap. */
   tokenBudget?: number
   tokenBudget?: number
-  /** Model context window in characters (wiki-store `maxContextSize`). */
+  /** Model context window in tokens (wiki-store `maxContextSize`). */
   maxContextSize?: number
   maxContextSize?: number
 }
 }
 export interface ComposedContext {
 export interface ComposedContext {

+ 1 - 1
src/lib/context-hub/types.ts

@@ -141,7 +141,7 @@ export interface ContextHubRequest {
   existingSummary?: SessionContextSummary
   existingSummary?: SessionContextSummary
   /** Explicit token budget; 0 / undefined = window-derived safe cap. */
   /** Explicit token budget; 0 / undefined = window-derived safe cap. */
   tokenBudget?: number
   tokenBudget?: number
-  /** Model context window in characters (wiki-store `maxContextSize`). */
+  /** Model context window in tokens (wiki-store `maxContextSize`). */
   maxContextSize?: number
   maxContextSize?: number
   forceRefresh?: boolean
   forceRefresh?: boolean
 }
 }

+ 13 - 1
src/lib/env-llm-defaults.ts

@@ -1,4 +1,8 @@
 import type { LlmConfig, ProviderConfigs } from "@/stores/wiki-store"
 import type { LlmConfig, ProviderConfigs } from "@/stores/wiki-store"
+import {
+  normalizeUserLlmContextSize,
+  normalizeUserLlmMaxOutputTokens,
+} from "@/lib/llm-context-size"
 
 
 const trimEnv = (value: unknown): string => {
 const trimEnv = (value: unknown): string => {
   return typeof value === "string" ? value.trim() : ""
   return typeof value === "string" ? value.trim() : ""
@@ -6,7 +10,12 @@ const trimEnv = (value: unknown): string => {
 
 
 const readContextSize = (): number => {
 const readContextSize = (): number => {
   const raw = Number(trimEnv(import.meta.env.VITE_QMAI_LLM_CONTEXT_SIZE))
   const raw = Number(trimEnv(import.meta.env.VITE_QMAI_LLM_CONTEXT_SIZE))
-  return Number.isFinite(raw) && raw > 0 ? raw : 204800
+  return normalizeUserLlmContextSize(raw)
+}
+
+const readMaxOutputTokens = (): number => {
+  const raw = Number(trimEnv(import.meta.env.VITE_QMAI_LLM_MAX_OUTPUT_TOKENS))
+  return normalizeUserLlmMaxOutputTokens(raw)
 }
 }
 
 
 export function loadEnvLlmDefault(): {
 export function loadEnvLlmDefault(): {
@@ -21,6 +30,7 @@ export function loadEnvLlmDefault(): {
   if (!apiKey || !customEndpoint || !model) return null
   if (!apiKey || !customEndpoint || !model) return null
 
 
   const maxContextSize = readContextSize()
   const maxContextSize = readContextSize()
+  const maxOutputTokens = readMaxOutputTokens()
   const config: LlmConfig = {
   const config: LlmConfig = {
     provider: "custom",
     provider: "custom",
     apiKey,
     apiKey,
@@ -28,6 +38,7 @@ export function loadEnvLlmDefault(): {
     ollamaUrl: "http://localhost:11434",
     ollamaUrl: "http://localhost:11434",
     customEndpoint,
     customEndpoint,
     maxContextSize,
     maxContextSize,
+    maxOutputTokens,
     apiMode: "chat_completions",
     apiMode: "chat_completions",
     reasoning: { mode: "auto" },
     reasoning: { mode: "auto" },
   }
   }
@@ -41,6 +52,7 @@ export function loadEnvLlmDefault(): {
         baseUrl: customEndpoint,
         baseUrl: customEndpoint,
         apiMode: "chat_completions",
         apiMode: "chat_completions",
         maxContextSize,
         maxContextSize,
+        maxOutputTokens,
         reasoning: { mode: "auto" },
         reasoning: { mode: "auto" },
       },
       },
     },
     },

+ 22 - 21
src/lib/ingest.prompt.test.ts

@@ -8,28 +8,29 @@ import {
   splitSourceIntoSemanticChunks,
   splitSourceIntoSemanticChunks,
 } from "./ingest"
 } from "./ingest"
 
 
-// langScale=1 pins these ladder-math tests to the English window so they
-// stay deterministic regardless of the active UI language (default zh).
+// The character-domain helpers take chars/token explicitly (4 = English-ish,
+// 1 = CJK) so they stay deterministic regardless of the active UI language.
 describe("long-source ingest planning", () => {
 describe("long-source ingest planning", () => {
   it("scales generation output tokens with the configured context window", () => {
   it("scales generation output tokens with the configured context window", () => {
-    expect(computeIngestGenerationMaxTokens(64_000, 1)).toBe(8_192)
-    expect(computeIngestGenerationMaxTokens(128_000, 1)).toBe(16_384)
-    expect(computeIngestGenerationMaxTokens(256_000, 1)).toBe(24_576)
-    expect(computeIngestGenerationMaxTokens(1_000_000, 1)).toBe(32_768)
-    expect(computeIngestReviewMaxTokens(1_000_000, 1)).toBe(8_192)
+    expect(computeIngestGenerationMaxTokens(64_000)).toBe(8_192)
+    expect(computeIngestGenerationMaxTokens(128_000)).toBe(16_384)
+    expect(computeIngestGenerationMaxTokens(256_000)).toBe(24_576)
+    expect(computeIngestGenerationMaxTokens(1_000_000)).toBe(32_768)
+    expect(computeIngestReviewMaxTokens(1_000_000)).toBe(8_192)
   })
   })
 
 
-  it("drops to a lower output tier under CJK scaling for the same window", () => {
-    // 128000 chars * 0.425 ≈ 54400 → below the 128K tier → default 8192.
-    expect(computeIngestGenerationMaxTokens(128_000, 0.425)).toBe(8_192)
+  it("picks the output tier from the token window alone", () => {
+    // A model's output ceiling is a property of the model, not of the UI
+    // language, so the tier no longer moves with character density.
+    expect(computeIngestGenerationMaxTokens(128_000)).toBe(16_384)
   })
   })
 
 
   it("scales analysis output tokens with the window but caps at 8192 (floor 4096)", () => {
   it("scales analysis output tokens with the window but caps at 8192 (floor 4096)", () => {
     // Small window keeps the legacy 4096 floor.
     // Small window keeps the legacy 4096 floor.
-    expect(computeIngestAnalysisMaxTokens(64_000, 1)).toBe(4_096)
+    expect(computeIngestAnalysisMaxTokens(64_000)).toBe(4_096)
     // Larger windows scale up but never exceed the 8192 cap.
     // Larger windows scale up but never exceed the 8192 cap.
-    expect(computeIngestAnalysisMaxTokens(128_000, 1)).toBe(8_192)
-    expect(computeIngestAnalysisMaxTokens(1_000_000, 1)).toBe(8_192)
+    expect(computeIngestAnalysisMaxTokens(128_000)).toBe(8_192)
+    expect(computeIngestAnalysisMaxTokens(1_000_000)).toBe(8_192)
   })
   })
 
 
   it("scales source budget from the configured context window instead of a fixed 50k cap", () => {
   it("scales source budget from the configured context window instead of a fixed 50k cap", () => {
@@ -41,9 +42,9 @@ describe("long-source ingest planning", () => {
     expect(large).toBeLessThanOrEqual(300_000)
     expect(large).toBeLessThanOrEqual(300_000)
   })
   })
 
 
-  it("shrinks the source budget under CJK scaling", () => {
-    const en = computeIngestSourceBudget(1_000_000, 8_000, 1)
-    const zh = computeIngestSourceBudget(1_000_000, 8_000, 0.425)
+  it("gives CJK fewer characters than English for the same token window", () => {
+    const en = computeIngestSourceBudget(200_000, 8_000, 4)
+    const zh = computeIngestSourceBudget(200_000, 8_000, 1)
     expect(zh).toBeLessThan(en)
     expect(zh).toBeLessThan(en)
   })
   })
 
 
@@ -52,9 +53,9 @@ describe("long-source ingest planning", () => {
   })
   })
 
 
   it("shrinks output tokens so prompt + output fits the window", () => {
   it("shrinks output tokens so prompt + output fits the window", () => {
-    // 64000-char window → 16000 tokens; 60000-char prompt → 15000 tokens in;
-    // only 1000 tokens left for output.
-    expect(fitIngestOutputToWindow(64_000, 60_000, 8_192, 1)).toBe(1_000)
+    // 64000-token window; a 60000-character CJK prompt is 60000 tokens in,
+    // leaving 4000 for output.
+    expect(fitIngestOutputToWindow(64_000, 60_000, 8_192, 1)).toBe(4_000)
   })
   })
 
 
   it("falls back to the output floor when the prompt already overflows", () => {
   it("falls back to the output floor when the prompt already overflows", () => {
@@ -62,8 +63,8 @@ describe("long-source ingest planning", () => {
   })
   })
 
 
   it("leaves less output room for CJK prompts than English ones", () => {
   it("leaves less output room for CJK prompts than English ones", () => {
-    const en = fitIngestOutputToWindow(64_000, 40_000, 8_192, 1)
-    const zh = fitIngestOutputToWindow(64_000, 40_000, 8_192, 0.425)
+    const en = fitIngestOutputToWindow(64_000, 250_000, 8_192, 4)
+    const zh = fitIngestOutputToWindow(64_000, 250_000, 8_192, 1)
     expect(zh).toBeLessThan(en)
     expect(zh).toBeLessThan(en)
   })
   })
 
 

+ 25 - 18
src/lib/ingest.ts

@@ -1220,9 +1220,9 @@ function clampNumber(value: number, min: number, max: number): number {
 export function computeIngestSourceBudget(
 export function computeIngestSourceBudget(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
   stableContextLength: number,
   stableContextLength: number,
-  langScale?: number,
+  charsPerToken?: number,
 ): number {
 ): number {
-  const { maxCtx, responseReserve } = computeContextBudget(maxContextSize, langScale)
+  const { maxCtx, responseReserve } = computeContextBudget(maxContextSize, charsPerToken)
   const stableReserve = Math.min(Math.floor(maxCtx * 0.25), Math.max(12_000, stableContextLength))
   const stableReserve = Math.min(Math.floor(maxCtx * 0.25), Math.max(12_000, stableContextLength))
   const instructionReserve = Math.max(12_000, Math.floor(maxCtx * 0.08))
   const instructionReserve = Math.max(12_000, Math.floor(maxCtx * 0.08))
   const available = maxCtx - responseReserve - stableReserve - instructionReserve
   const available = maxCtx - responseReserve - stableReserve - instructionReserve
@@ -1230,22 +1230,28 @@ export function computeIngestSourceBudget(
   return clampNumber(Math.floor(available), LONG_SOURCE_MIN_BUDGET, upper)
   return clampNumber(Math.floor(available), LONG_SOURCE_MIN_BUDGET, upper)
 }
 }
 
 
+/**
+ * Output ladder for wiki page generation, stepped off the model's token
+ * window. Compares the window directly rather than a language-scaled
+ * character budget: a model's output ceiling does not shrink because the UI
+ * is in Chinese, and the old comparison dropped CJK users a whole tier.
+ */
 export function computeIngestGenerationMaxTokens(
 export function computeIngestGenerationMaxTokens(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
-  langScale?: number,
 ): number {
 ): number {
-  const { maxCtx } = computeContextBudget(maxContextSize, langScale)
-  if (maxCtx >= 512_000) return INGEST_GENERATION_TOKENS_512K
-  if (maxCtx >= 256_000) return INGEST_GENERATION_TOKENS_256K
-  if (maxCtx >= 128_000) return INGEST_GENERATION_TOKENS_128K
+  const windowTokens = typeof maxContextSize === "number" && maxContextSize > 0
+    ? maxContextSize
+    : DEFAULT_INGEST_WINDOW_TOKENS
+  if (windowTokens >= 512_000) return INGEST_GENERATION_TOKENS_512K
+  if (windowTokens >= 256_000) return INGEST_GENERATION_TOKENS_256K
+  if (windowTokens >= 128_000) return INGEST_GENERATION_TOKENS_128K
   return INGEST_GENERATION_TOKENS_DEFAULT
   return INGEST_GENERATION_TOKENS_DEFAULT
 }
 }
 
 
 export function computeIngestReviewMaxTokens(
 export function computeIngestReviewMaxTokens(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
-  langScale?: number,
 ): number {
 ): number {
-  return Math.min(8_192, Math.max(4_096, Math.floor(computeIngestGenerationMaxTokens(maxContextSize, langScale) / 2)))
+  return Math.min(8_192, Math.max(4_096, Math.floor(computeIngestGenerationMaxTokens(maxContextSize) / 2)))
 }
 }
 
 
 /**
 /**
@@ -1257,13 +1263,14 @@ export function computeIngestReviewMaxTokens(
  */
  */
 export function computeIngestAnalysisMaxTokens(
 export function computeIngestAnalysisMaxTokens(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
-  langScale?: number,
 ): number {
 ): number {
-  return Math.min(8_192, Math.max(4_096, Math.floor(computeIngestGenerationMaxTokens(maxContextSize, langScale) / 2)))
+  return Math.min(8_192, Math.max(4_096, Math.floor(computeIngestGenerationMaxTokens(maxContextSize) / 2)))
 }
 }
 
 
 /** chars/token the ingest budgeting assumes; mirrors context-budget.ts. */
 /** chars/token the ingest budgeting assumes; mirrors context-budget.ts. */
 const INGEST_CHARS_PER_TOKEN = 4
 const INGEST_CHARS_PER_TOKEN = 4
+/** Window assumed when the config carries none; mirrors context-budget.ts. */
+const DEFAULT_INGEST_WINDOW_TOKENS = 204_800
 /** Smallest output allowance we will still request when the window is nearly
 /** Smallest output allowance we will still request when the window is nearly
  *  full — below this a response is useless, so we accept a tiny overflow risk
  *  full — below this a response is useless, so we accept a tiny overflow risk
  *  rather than emitting nothing. */
  *  rather than emitting nothing. */
@@ -1274,19 +1281,19 @@ const INGEST_OUTPUT_TOKEN_FLOOR = 512
  * model's real token window. `desiredTokens` is the ladder value; we only ever
  * model's real token window. `desiredTokens` is the ladder value; we only ever
  * reduce it when the prompt already leaves less room than the ladder wants.
  * reduce it when the prompt already leaves less room than the ladder wants.
  *
  *
- * Language-aware: CJK text is ~2.3x denser, so the same prompt consumes more
- * real tokens and leaves less room for output. The raw (unscaled) window is
- * the real token capacity (English-calibrated 4:1); the effective scale
- * recovers the true chars/token for the active language.
+ * Language-aware: CJK text is denser, so the same prompt consumes more real
+ * tokens and leaves less room for output. The English-calibrated window (4:1)
+ * recovers the real token capacity; the ratio against the active language's
+ * window recovers that language's true chars/token.
  */
  */
 export function fitIngestOutputToWindow(
 export function fitIngestOutputToWindow(
   maxContextSize: number | undefined,
   maxContextSize: number | undefined,
   promptChars: number,
   promptChars: number,
   desiredTokens: number,
   desiredTokens: number,
-  langScale?: number,
+  charsPerToken?: number,
 ): number {
 ): number {
-  const rawWindow = computeContextBudget(maxContextSize, 1).maxCtx
-  const scaledWindow = computeContextBudget(maxContextSize, langScale).maxCtx
+  const rawWindow = computeContextBudget(maxContextSize, INGEST_CHARS_PER_TOKEN).maxCtx
+  const scaledWindow = computeContextBudget(maxContextSize, charsPerToken).maxCtx
   const scale = rawWindow > 0 ? scaledWindow / rawWindow : 1
   const scale = rawWindow > 0 ? scaledWindow / rawWindow : 1
   const windowTokens = rawWindow / INGEST_CHARS_PER_TOKEN
   const windowTokens = rawWindow / INGEST_CHARS_PER_TOKEN
   const inputTokens = promptChars / (INGEST_CHARS_PER_TOKEN * scale)
   const inputTokens = promptChars / (INGEST_CHARS_PER_TOKEN * scale)

+ 87 - 15
src/lib/llm-client.ts

@@ -2,8 +2,10 @@ import type { LlmConfig } from "@/stores/wiki-store"
 import { isAzureOpenAiEndpoint } from "@/lib/azure-openai"
 import { isAzureOpenAiEndpoint } from "@/lib/azure-openai"
 import {
 import {
   getEffectiveMaxContextSize,
   getEffectiveMaxContextSize,
+  getEffectiveMaxOutputTokens,
   getProviderConfig,
   getProviderConfig,
   isTruncationFinishReason,
   isTruncationFinishReason,
+  thinkingMinMaxTokens,
   type RequestOverrides,
   type RequestOverrides,
 } from "./llm-providers"
 } from "./llm-providers"
 import { getHttpFetch, isFetchNetworkError } from "./tauri-fetch"
 import { getHttpFetch, isFetchNetworkError } from "./tauri-fetch"
@@ -16,7 +18,12 @@ import {
 } from "./reasoning-replay-debug"
 } from "./reasoning-replay-debug"
 import { resolveRuntimeLocalCliConfig } from "./local-cli-config"
 import { resolveRuntimeLocalCliConfig } from "./local-cli-config"
 import { ensureCursorProxyRunning, withCursorProxyEndpoint } from "./cursor-cli-proxy"
 import { ensureCursorProxyRunning, withCursorProxyEndpoint } from "./cursor-cli-proxy"
-import { trimChatMessagesToBudget } from "./chat-request-budget"
+import {
+  estimateChatMessagesTokens,
+  estimateRequestScaffoldTokens,
+  trimChatMessagesToTokenBudget,
+} from "./chat-request-budget"
+import { RESPONSE_RESERVE_FRAC, planLlmRequestBudget } from "./context-budget"
 import { mergeLlmUsageSnapshot, type LlmUsage } from "./llm-usage"
 import { mergeLlmUsageSnapshot, type LlmUsage } from "./llm-usage"
 import { applyGlobalUserMemoryToMessages } from "./user-memory/request-integration"
 import { applyGlobalUserMemoryToMessages } from "./user-memory/request-integration"
 
 
@@ -131,6 +138,8 @@ function parseToolCallDeltaFromLine(line: string): { index: number; id?: string;
       arguments: toolCall.function?.arguments,
       arguments: toolCall.function?.arguments,
     }
     }
   } catch {
   } catch {
+    // A malformed SSE line is not fatal: skip it and keep the stream alive.
+    // The only error reachable here is JSON.parse's SyntaxError.
     return null
     return null
   }
   }
 }
 }
@@ -166,16 +175,62 @@ export async function streamChat(
 ): Promise<void> {
 ): Promise<void> {
   let runtimeConfig = await resolveRuntimeLocalCliConfig(config)
   let runtimeConfig = await resolveRuntimeLocalCliConfig(config)
   const preparedMessages = applyGlobalUserMemoryToMessages(messages, requestOverrides)
   const preparedMessages = applyGlobalUserMemoryToMessages(messages, requestOverrides)
-  // Apply model-specific context size minimums (e.g. DeepSeek → 1M)
   const configuredWindow = getEffectiveMaxContextSize(runtimeConfig)
   const configuredWindow = getEffectiveMaxContextSize(runtimeConfig)
-  const outputReserveChars = requestOverrides?.max_tokens
-    ? Math.max(0, requestOverrides.max_tokens * 4)
-    : Math.floor(configuredWindow * 0.15)
-  const requestInputBudget = Math.max(1, Math.min(
-    Math.floor(configuredWindow * 0.85),
-    configuredWindow - outputReserveChars,
-  ))
-  const budgetedMessages = trimChatMessagesToBudget(preparedMessages, requestInputBudget)
+  const toolScaffoldTokens = estimateRequestScaffoldTokens(requestOverrides?.tools)
+  const outputCap = getEffectiveMaxOutputTokens(runtimeConfig)
+  const thinkingFloorTokens = thinkingMinMaxTokens(runtimeConfig.reasoning ?? { mode: "auto" })
+  const runtimeBudget = planLlmRequestBudget({
+    maxContextSize: configuredWindow,
+    // Without an explicit request the response reserve is only used to size
+    // the INPUT trim; it is not sent as max_tokens unless thinking needs it
+    // (see shouldSendMaxTokens below).
+    desiredOutputTokens: requestOverrides?.max_tokens
+      ?? Math.floor(configuredWindow * RESPONSE_RESERVE_FRAC),
+    scaffoldReserveTokens: toolScaffoldTokens,
+    minimumContextTokens: 64,
+    maxOutputTokensCap: outputCap,
+    thinkingFloorTokens,
+  })
+  let effectiveOutputTokens = runtimeBudget.outputTokens
+  // Emit max_tokens when the caller asked for one, or when explicit thinking
+  // needs a known output allowance (otherwise OpenAI-compatible paths keep
+  // thinking on against an unknown provider default). auto/off without a
+  // caller value still omits the field so long-form keeps the provider default.
+  let shouldSendMaxTokens =
+    requestOverrides?.max_tokens !== undefined || thinkingFloorTokens > 0
+  let budgetedMessages: import("./llm-providers").ChatMessage[]
+  try {
+    budgetedMessages = trimChatMessagesToTokenBudget(
+      preparedMessages,
+      runtimeBudget.inputTokenBudget - toolScaffoldTokens,
+    )
+  } catch {
+    // Protected system/current-user content did not fit beside the desired output.
+    // Retry locally with the 512-token floor; no provider request is made on failure.
+    const minimumOutputTokens = 512
+    const maximumInputTokens = runtimeBudget.windowTokens
+      - toolScaffoldTokens
+      - minimumOutputTokens
+    budgetedMessages = trimChatMessagesToTokenBudget(
+      preparedMessages,
+      maximumInputTokens,
+    )
+    effectiveOutputTokens = Math.max(
+      minimumOutputTokens,
+      Math.min(
+        runtimeBudget.outputTokens,
+        runtimeBudget.windowTokens
+          - toolScaffoldTokens
+          - estimateChatMessagesTokens(budgetedMessages),
+      ),
+    )
+    // The input was trimmed against a reserved output slot, so that slot has
+    // to be declared even if the caller never asked for one.
+    shouldSendMaxTokens = true
+  }
+  const effectiveRequestOverrides: RequestOverrides = shouldSendMaxTokens
+    ? { ...requestOverrides, max_tokens: effectiveOutputTokens }
+    : { ...requestOverrides }
   const { onToken, onDone, onError } = callbacks
   const { onToken, onDone, onError } = callbacks
   const decoder = new TextDecoder()
   const decoder = new TextDecoder()
 
 
@@ -183,11 +238,11 @@ export async function streamChat(
   // HTTP. Dispatch before getProviderConfig — that function throws for
   // HTTP. Dispatch before getProviderConfig — that function throws for
   // this provider because it has no URL/headers.
   // this provider because it has no URL/headers.
   if (runtimeConfig.provider === "claude-code") {
   if (runtimeConfig.provider === "claude-code") {
-    return streamViaClaudeCodeCli(runtimeConfig, budgetedMessages, callbacks, signal, requestOverrides)
+    return streamViaClaudeCodeCli(runtimeConfig, budgetedMessages, callbacks, signal, effectiveRequestOverrides)
   }
   }
 
 
   if (runtimeConfig.provider === "codex-cli") {
   if (runtimeConfig.provider === "codex-cli") {
-    return streamViaCodexCli(runtimeConfig, budgetedMessages, callbacks, signal, requestOverrides)
+    return streamViaCodexCli(runtimeConfig, budgetedMessages, callbacks, signal, effectiveRequestOverrides)
   }
   }
 
 
   if (runtimeConfig.provider === "cursor-cli") {
   if (runtimeConfig.provider === "cursor-cli") {
@@ -230,7 +285,7 @@ export async function streamChat(
     const buildRequestInit = (nextMessages: import("./llm-providers").ChatMessage[]): RequestInit => ({
     const buildRequestInit = (nextMessages: import("./llm-providers").ChatMessage[]): RequestInit => ({
       method: "POST",
       method: "POST",
       headers: providerConfig.headers,
       headers: providerConfig.headers,
-      body: JSON.stringify(providerConfig.buildBody(nextMessages, requestOverrides)),
+      body: JSON.stringify(providerConfig.buildBody(nextMessages, effectiveRequestOverrides)),
       signal: combinedSignal,
       signal: combinedSignal,
     })
     })
 
 
@@ -327,9 +382,26 @@ export async function streamChat(
       let inputLimitRetrySucceeded = false
       let inputLimitRetrySucceeded = false
       const inputLimit = parseInputLengthLimit(errorDetail)
       const inputLimit = parseInputLengthLimit(errorDetail)
       if (inputLimit) {
       if (inputLimit) {
-        const retryRequestInit = buildRequestInit(
-          trimChatMessagesToBudget(budgetedMessages, Math.floor(inputLimit.maxLength * 0.85)),
+        const currentInputTokens = estimateChatMessagesTokens(budgetedMessages)
+        // The provider reports the overshoot in characters; we trim in tokens.
+        // Applying the ratio across units is a heuristic, not an exact
+        // conversion — it only has to land us under the limit, and the 0.85
+        // factor absorbs the imprecision.
+        const shrinkRatio = Math.min(1, inputLimit.maxLength / Math.max(1, inputLimit.inputLength))
+        const retryInputTokenBudget = Math.max(
+          1,
+          Math.floor(currentInputTokens * shrinkRatio * 0.85),
         )
         )
+        let retryMessages: import("./llm-providers").ChatMessage[]
+        try {
+          retryMessages = trimChatMessagesToTokenBudget(budgetedMessages, retryInputTokenBudget)
+        } catch {
+          // Even the protected messages exceed the provider's limit; there is
+          // nothing left to shrink, so report the original limit.
+          onError(new Error(inputLengthLimitMessage(inputLimit)))
+          return
+        }
+        const retryRequestInit = buildRequestInit(retryMessages)
         if (retryRequestInit.body === requestInit.body) {
         if (retryRequestInit.body === requestInit.body) {
           onError(new Error(inputLengthLimitMessage(inputLimit)))
           onError(new Error(inputLengthLimitMessage(inputLimit)))
           return
           return

+ 148 - 5
src/lib/llm-client.usage.spec.ts

@@ -1,6 +1,15 @@
 import { beforeEach, describe, expect, it, vi } from "vitest"
 import { beforeEach, describe, expect, it, vi } from "vitest"
 import type { LlmConfig } from "@/stores/wiki-store"
 import type { LlmConfig } from "@/stores/wiki-store"
 import { streamChat } from "./llm-client"
 import { streamChat } from "./llm-client"
+import { estimateChatMessagesTokens } from "./chat-request-budget"
+import type { ChatMessage } from "./llm-providers"
+import { thinkingMinMaxTokens } from "./llm-providers"
+import {
+  LlmContextBudgetError,
+  RESPONSE_RESERVE_FRAC,
+  planLlmRequestBudget,
+} from "./context-budget"
+import { normalizeUserLlmMaxOutputTokens } from "./llm-context-size"
 
 
 const mocks = vi.hoisted(() => ({
 const mocks = vi.hoisted(() => ({
   fetch: vi.fn(),
   fetch: vi.fn(),
@@ -101,14 +110,16 @@ describe("streamChat usage", () => {
     }))
     }))
   })
   })
 
 
-  it("发送前把总输入限制在模型窗口的 85%", async () => {
+  it("发送前按 token 预算裁剪并保持系统与当前请求非空", async () => {
     mocks.fetch.mockResolvedValue(new Response([
     mocks.fetch.mockResolvedValue(new Response([
       'data: {"choices":[{"delta":{"content":"完成"}}]}',
       'data: {"choices":[{"delta":{"content":"完成"}}]}',
       "data: [DONE]",
       "data: [DONE]",
       "",
       "",
     ].join("\n"), { status: 200 }))
     ].join("\n"), { status: 200 }))
 
 
-    await streamChat({ ...config, maxContextSize: 1_000 }, [
+    // 1843-token window (2048 × 0.9) against ~1800 tokens of CJK input, so the
+    // trim has to bite while leaving the protected messages intact.
+    await streamChat({ ...config, maxContextSize: 2_048 }, [
       { role: "system", content: "系统".repeat(450) },
       { role: "system", content: "系统".repeat(450) },
       { role: "user", content: `任务目标:续写。${"正文".repeat(450)}结尾限制:保持人物关系。` },
       { role: "user", content: `任务目标:续写。${"正文".repeat(450)}结尾限制:保持人物关系。` },
     ], {
     ], {
@@ -118,10 +129,142 @@ describe("streamChat usage", () => {
     })
     })
 
 
     const request = mocks.fetch.mock.calls[0][1] as RequestInit
     const request = mocks.fetch.mock.calls[0][1] as RequestInit
-    const body = JSON.parse(String(request.body)) as { messages: Array<{ content: string }> }
-    const total = body.messages.reduce((sum, message) => sum + message.content.length, 0)
-    expect(total).toBeLessThanOrEqual(850)
+    const body = JSON.parse(String(request.body)) as {
+      messages: ChatMessage[]
+      max_tokens?: number
+    }
+    expect(estimateChatMessagesTokens(body.messages)).toBeLessThanOrEqual(1_331)
+    expect(String(body.messages[0]?.content).trim()).not.toBe("")
     expect(body.messages.at(-1)?.content).toContain("任务目标")
     expect(body.messages.at(-1)?.content).toContain("任务目标")
     expect(body.messages.at(-1)?.content).toContain("保持人物关系")
     expect(body.messages.at(-1)?.content).toContain("保持人物关系")
   })
   })
+
+  it("上下文无法容纳最小输出时明确失败且不调用供应商", async () => {
+    await expect(streamChat({ ...config, maxContextSize: 512 }, [
+      { role: "system", content: "系统约束" },
+      { role: "user", content: "生成第一卷完整大纲" },
+    ], {
+      onToken: vi.fn(),
+      onDone: vi.fn(),
+      onError: vi.fn(),
+    })).rejects.toBeInstanceOf(LlmContextBudgetError)
+
+    expect(mocks.fetch).not.toHaveBeenCalled()
+  })
+
+  it("调用方未传 max_tokens 时请求体不带该字段", async () => {
+    mocks.fetch.mockResolvedValue(new Response([
+      'data: {"choices":[{"delta":{"content":"完成"}}]}',
+      "data: [DONE]",
+      "",
+    ].join("\n"), { status: 200 }))
+
+    await streamChat(config, [{ role: "user", content: "写第一章" }], {
+      onToken: vi.fn(),
+      onDone: vi.fn(),
+      onError: vi.fn(),
+    })
+
+    const request = mocks.fetch.mock.calls[0][1] as RequestInit
+    expect(JSON.parse(String(request.body))).not.toHaveProperty("max_tokens")
+  })
+
+  it("reasoning.mode=auto 且调用方未传 max_tokens 时请求体仍省略该字段", async () => {
+    mocks.fetch.mockResolvedValue(new Response([
+      'data: {"choices":[{"delta":{"content":"完成"}}]}',
+      "data: [DONE]",
+      "",
+    ].join("\n"), { status: 200 }))
+
+    await streamChat(
+      { ...config, reasoning: { mode: "auto" } },
+      [{ role: "user", content: "写第一章" }],
+      { onToken: vi.fn(), onDone: vi.fn(), onError: vi.fn() },
+    )
+
+    const request = mocks.fetch.mock.calls[0][1] as RequestInit
+    expect(JSON.parse(String(request.body))).not.toHaveProperty("max_tokens")
+  })
+
+  it("reasoning.mode=high 且调用方未传 max_tokens 时发送预算规划的 max_tokens", async () => {
+    mocks.fetch.mockResolvedValue(new Response([
+      'data: {"choices":[{"delta":{"content":"完成"}}]}',
+      "data: [DONE]",
+      "",
+    ].join("\n"), { status: 200 }))
+
+    const reasoning = { mode: "high" as const }
+    const thinkingFloorTokens = thinkingMinMaxTokens(reasoning)
+    expect(thinkingFloorTokens).toBeGreaterThan(0)
+    const planned = planLlmRequestBudget({
+      maxContextSize: config.maxContextSize,
+      desiredOutputTokens: Math.floor(config.maxContextSize * RESPONSE_RESERVE_FRAC),
+      scaffoldReserveTokens: 0,
+      minimumContextTokens: 64,
+      maxOutputTokensCap: normalizeUserLlmMaxOutputTokens(config.maxOutputTokens),
+      thinkingFloorTokens,
+    })
+
+    await streamChat(
+      { ...config, reasoning },
+      [{ role: "user", content: "写第一章" }],
+      { onToken: vi.fn(), onDone: vi.fn(), onError: vi.fn() },
+    )
+
+    const request = mocks.fetch.mock.calls[0][1] as RequestInit
+    expect(JSON.parse(String(request.body))).toMatchObject({
+      max_tokens: planned.outputTokens,
+    })
+    expect(planned.outputTokens).toBeGreaterThanOrEqual(thinkingFloorTokens)
+  })
+
+  it("调用方显式传入的超大 max_tokens 收敛到输出上限", async () => {
+    mocks.fetch.mockResolvedValue(new Response([
+      'data: {"choices":[{"delta":{"content":"完成"}}]}',
+      "data: [DONE]",
+      "",
+    ].join("\n"), { status: 200 }))
+
+    await streamChat(
+      { ...config, maxContextSize: 1_000_000, maxOutputTokens: 65_536 },
+      [{ role: "user", content: "写第一章" }],
+      { onToken: vi.fn(), onDone: vi.fn(), onError: vi.fn() },
+      undefined,
+      { max_tokens: 300_000 },
+    )
+
+    const request = mocks.fetch.mock.calls[0][1] as RequestInit
+    expect(JSON.parse(String(request.body))).toMatchObject({ max_tokens: 65_536 })
+  })
+
+  it("脏 SSE 行不会中断整轮流式响应", async () => {
+    const encoder = new TextEncoder()
+    const body = new ReadableStream<Uint8Array>({
+      start(controller) {
+        controller.enqueue(encoder.encode([
+          'data: {"choices":[{"delta":{"content":"前半"}}]}',
+          "data: {不是合法 JSON",
+          'data: {"choices":[{"delta":{"content":"后半"}}]}',
+          "data: [DONE]",
+          "",
+        ].join("\n")))
+        controller.close()
+      },
+    })
+    mocks.fetch.mockResolvedValue(new Response(body, { status: 200 }))
+    const onToken = vi.fn()
+    const onDone = vi.fn()
+    const onError = vi.fn()
+
+    await streamChat(config, [{ role: "user", content: "写第一章" }], {
+      onToken,
+      onDone,
+      onError,
+    })
+
+    expect(onToken).toHaveBeenCalledWith("前半")
+    expect(onToken).toHaveBeenCalledWith("后半")
+    expect(onDone).toHaveBeenCalledOnce()
+    expect(onError).not.toHaveBeenCalled()
+  })
 })
 })

+ 72 - 0
src/lib/llm-context-size.ts

@@ -0,0 +1,72 @@
+import type { LlmConfig, ProviderConfigs, ProviderOverride } from "@/stores/wiki-store"
+
+export const MIN_USER_LLM_CONTEXT_SIZE = 204_800
+
+/** Default declared output ceiling when neither the user nor the preset says
+ *  otherwise. Generous on purpose — it must not silently truncate capable
+ *  models — so presets should carry a real figure wherever one is known. */
+export const DEFAULT_USER_LLM_MAX_OUTPUT_TOKENS = 131_072
+/** Below this an answer is not worth requesting. */
+export const MIN_USER_LLM_MAX_OUTPUT_TOKENS = 512
+/** Highest output any model in the catalog declares (DeepSeek V4: 384K). */
+export const MAX_USER_LLM_MAX_OUTPUT_TOKENS = 393_216
+
+export function normalizeUserLlmContextSize(value: number | undefined): number {
+  if (!Number.isFinite(value) || (value as number) <= 0) {
+    return MIN_USER_LLM_CONTEXT_SIZE
+  }
+  return Math.max(MIN_USER_LLM_CONTEXT_SIZE, Math.floor(value as number))
+}
+
+/**
+ * Unlike the context window this has no floor, only a default: a user must be
+ * able to declare a small ceiling for a model that really does cap out low.
+ */
+export function normalizeUserLlmMaxOutputTokens(value: number | undefined): number {
+  if (!Number.isFinite(value) || (value as number) <= 0) {
+    return DEFAULT_USER_LLM_MAX_OUTPUT_TOKENS
+  }
+  return Math.max(
+    MIN_USER_LLM_MAX_OUTPUT_TOKENS,
+    Math.min(MAX_USER_LLM_MAX_OUTPUT_TOKENS, Math.floor(value as number)),
+  )
+}
+
+export function normalizeUserLlmConfig(config: LlmConfig): LlmConfig {
+  const maxContextSize = normalizeUserLlmContextSize(config.maxContextSize)
+  const maxOutputTokens = config.maxOutputTokens === undefined
+    ? undefined
+    : normalizeUserLlmMaxOutputTokens(config.maxOutputTokens)
+  return maxContextSize === config.maxContextSize && maxOutputTokens === config.maxOutputTokens
+    ? config
+    : { ...config, maxContextSize, ...(maxOutputTokens === undefined ? {} : { maxOutputTokens }) }
+}
+
+export function normalizeProviderOverride(override: ProviderOverride): ProviderOverride {
+  const maxContextSize = override.maxContextSize === undefined
+    ? undefined
+    : normalizeUserLlmContextSize(override.maxContextSize)
+  const maxOutputTokens = override.maxOutputTokens === undefined
+    ? undefined
+    : normalizeUserLlmMaxOutputTokens(override.maxOutputTokens)
+  if (maxContextSize === override.maxContextSize && maxOutputTokens === override.maxOutputTokens) {
+    return override
+  }
+  return {
+    ...override,
+    ...(maxContextSize === undefined ? {} : { maxContextSize }),
+    ...(maxOutputTokens === undefined ? {} : { maxOutputTokens }),
+  }
+}
+
+export function normalizeProviderConfigs(configs: ProviderConfigs): ProviderConfigs {
+  let changed = false
+  const normalized = Object.fromEntries(
+    Object.entries(configs).map(([id, override]) => {
+      const next = normalizeProviderOverride(override)
+      if (next !== override) changed = true
+      return [id, next]
+    }),
+  )
+  return changed ? normalized : configs
+}

+ 85 - 15
src/lib/llm-providers.spec.ts

@@ -23,6 +23,35 @@ function requestBody(config: LlmConfig): Record<string, unknown> {
 }
 }
 
 
 describe("llm provider reasoning options", () => {
 describe("llm provider reasoning options", () => {
+  it("keeps Anthropic thinking inside the caller's max_tokens budget", () => {
+    const body = getProviderConfig(customConfig({
+      apiMode: "anthropic_messages",
+      reasoning: { mode: "high" },
+    })).buildBody(
+      [{ role: "user", content: "请回答。" }],
+      { max_tokens: 4_096 },
+    ) as {
+      max_tokens: number
+      thinking?: { type: string; budget_tokens: number }
+    }
+
+    expect(body.max_tokens).toBe(4_096)
+    expect(body.thinking).toEqual({ type: "enabled", budget_tokens: 3_584 })
+  })
+
+  it("does not inflate an Anthropic output budget too small for explicit thinking", () => {
+    const body = getProviderConfig(customConfig({
+      apiMode: "anthropic_messages",
+      reasoning: { mode: "high" },
+    })).buildBody(
+      [{ role: "user", content: "请回答。" }],
+      { max_tokens: 512 },
+    ) as Record<string, unknown>
+
+    expect(body.max_tokens).toBe(512)
+    expect(body).not.toHaveProperty("thinking")
+  })
+
   it("replays assistant reasoning_content including empty string", () => {
   it("replays assistant reasoning_content including empty string", () => {
     const body = getProviderConfig(customConfig()).buildBody([
     const body = getProviderConfig(customConfig()).buildBody([
       { role: "user", content: "写第一章" },
       { role: "user", content: "写第一章" },
@@ -136,31 +165,28 @@ describe("llm provider reasoning options", () => {
     expect(body).not.toHaveProperty("thinking")
     expect(body).not.toHaveProperty("thinking")
   })
   })
 
 
-  it("boosts max_tokens for MiMo when thinking is enabled without explicit max_tokens", () => {
+  it("leaves max_tokens absent for MiMo thinking when the caller did not set one", () => {
     const body = requestBody(customConfig({
     const body = requestBody(customConfig({
       model: "mimo-v2.5-pro",
       model: "mimo-v2.5-pro",
       reasoning: { mode: "high" },
       reasoning: { mode: "high" },
     }))
     }))
 
 
-    expect(body.max_tokens).toBe(16384)
+    expect(body).not.toHaveProperty("max_tokens")
+    expect(body.chat_template_kwargs).toEqual({ enable_thinking: true })
   })
   })
 
 
-  it("boosts max_tokens for MiMo via endpoint detection", () => {
+  it("leaves max_tokens absent for MiMo detected by endpoint", () => {
     const body = requestBody(customConfig({
     const body = requestBody(customConfig({
       model: "custom-alias",
       model: "custom-alias",
       customEndpoint: "https://token-plan-cn.xiaomimimo.com/v1",
       customEndpoint: "https://token-plan-cn.xiaomimimo.com/v1",
       reasoning: { mode: "medium" },
       reasoning: { mode: "medium" },
     }))
     }))
 
 
-    expect(body.max_tokens).toBe(8192)
+    expect(body).not.toHaveProperty("max_tokens")
+    expect(body.chat_template_kwargs).toEqual({ enable_thinking: true })
   })
   })
 
 
-  it("does not override explicit larger max_tokens for MiMo thinking", () => {
-    const body = requestBody(customConfig({
-      model: "mimo-v2.5-pro",
-      reasoning: { mode: "high" },
-    }))
-    // Build body with explicit max_tokens override
+  it("never rewrites an explicit max_tokens for MiMo thinking", () => {
     const bodyWithOverride = getProviderConfig(customConfig({
     const bodyWithOverride = getProviderConfig(customConfig({
       model: "mimo-v2.5-pro",
       model: "mimo-v2.5-pro",
       reasoning: { mode: "high" },
       reasoning: { mode: "high" },
@@ -170,7 +196,36 @@ describe("llm provider reasoning options", () => {
     ) as Record<string, unknown>
     ) as Record<string, unknown>
 
 
     expect(bodyWithOverride.max_tokens).toBe(32000)
     expect(bodyWithOverride.max_tokens).toBe(32000)
-    expect(body.max_tokens).toBe(16384)
+    expect(bodyWithOverride.chat_template_kwargs).toEqual({ enable_thinking: true })
+  })
+
+  it("turns MiMo thinking off when the planned output cannot hold it", () => {
+    const body = getProviderConfig(customConfig({
+      model: "mimo-v2.5-pro",
+      reasoning: { mode: "high" },
+    })).buildBody(
+      [{ role: "user", content: "test" }],
+      { max_tokens: 2048 },
+    ) as Record<string, unknown>
+
+    expect(body.max_tokens).toBe(2048)
+    expect(body.chat_template_kwargs).toEqual({ enable_thinking: false })
+    expect(body).not.toHaveProperty("reasoning_effort")
+  })
+
+  it("turns GLM-5 thinking off when the planned output cannot hold it", () => {
+    const body = getProviderConfig(customConfig({
+      model: "glm-5-plus",
+      customEndpoint: "https://open.bigmodel.cn/api/paas/v4",
+      reasoning: { mode: "high" },
+    })).buildBody(
+      [{ role: "user", content: "test" }],
+      { max_tokens: 2048 },
+    ) as Record<string, unknown>
+
+    expect(body.max_tokens).toBe(2048)
+    expect(body.thinking).toEqual({ type: "disabled" })
+    expect(body).not.toHaveProperty("reasoning_effort")
   })
   })
 
 
   it("does not set max_tokens for MiMo when thinking is off", () => {
   it("does not set max_tokens for MiMo when thinking is off", () => {
@@ -191,23 +246,38 @@ describe("llm provider reasoning options", () => {
     expect(body).not.toHaveProperty("max_tokens")
     expect(body).not.toHaveProperty("max_tokens")
   })
   })
 
 
-  it("boosts max_tokens for Qwen3 thinking at medium level", () => {
+  it("leaves max_tokens absent for Qwen3 thinking at medium level", () => {
     const body = requestBody(customConfig({
     const body = requestBody(customConfig({
       model: "qwen3-235b-a22b",
       model: "qwen3-235b-a22b",
       reasoning: { mode: "medium" },
       reasoning: { mode: "medium" },
     }))
     }))
 
 
-    expect(body.max_tokens).toBe(8192)
+    expect(body).not.toHaveProperty("max_tokens")
+    expect(body.chat_template_kwargs).toEqual({ enable_thinking: true })
   })
   })
 
 
-  it("boosts max_tokens for DeepSeek thinking at low level", () => {
+  it("leaves max_tokens absent for DeepSeek thinking at low level", () => {
     const body = requestBody(customConfig({
     const body = requestBody(customConfig({
       model: "deepseek-v4-flash",
       model: "deepseek-v4-flash",
       reasoning: { mode: "low" },
       reasoning: { mode: "low" },
     }))
     }))
 
 
     expect(body.thinking).toEqual({ type: "enabled" })
     expect(body.thinking).toEqual({ type: "enabled" })
-    expect(body.max_tokens).toBe(4096)
+    expect(body).not.toHaveProperty("max_tokens")
+  })
+
+  it("turns DeepSeek thinking off when the planned output cannot hold it", () => {
+    const body = getProviderConfig(customConfig({
+      model: "deepseek-v4-flash",
+      reasoning: { mode: "high" },
+    })).buildBody(
+      [{ role: "user", content: "test" }],
+      { max_tokens: 2048 },
+    ) as Record<string, unknown>
+
+    expect(body.max_tokens).toBe(2048)
+    expect(body.thinking).toEqual({ type: "disabled" })
+    expect(body).not.toHaveProperty("reasoning_effort")
   })
   })
 
 
   it.each<ReasoningMode>(["max", "custom"])("maps Responses API %s reasoning to high effort", (mode) => {
   it.each<ReasoningMode>(["max", "custom"])("maps Responses API %s reasoning to high effort", (mode) => {

+ 73 - 40
src/lib/llm-providers.ts

@@ -5,6 +5,10 @@ import {
   isAzureOpenAiEndpoint,
   isAzureOpenAiEndpoint,
 } from "@/lib/azure-openai"
 } from "@/lib/azure-openai"
 import { normalizeEndpoint } from "@/lib/endpoint-normalizer"
 import { normalizeEndpoint } from "@/lib/endpoint-normalizer"
+import {
+  MIN_USER_LLM_CONTEXT_SIZE,
+  normalizeUserLlmMaxOutputTokens,
+} from "@/lib/llm-context-size"
 import type { LlmUsage } from "./llm-usage"
 import type { LlmUsage } from "./llm-usage"
 import type { UserMemorySurface } from "./user-memory/types"
 import type { UserMemorySurface } from "./user-memory/types"
 
 
@@ -603,16 +607,20 @@ function reasoningEffort(reasoning: ReasoningConfig): "low" | "medium" | "high"
 }
 }
 
 
 /**
 /**
- * Minimum total output tokens (thinking + final answer) required when
- * chain-of-thought is explicitly enabled.  Without this floor the API's
- * default `max_tokens` can be too small to hold both the reasoning trace
- * and the final content — the model spends every token on `reasoning_content`
- * and produces zero `content`, which surfaces as the "思考上限" error.
+ * Total output tokens (thinking + final answer) a reasoning level needs in
+ * order to produce anything useful. Below this the model spends the whole
+ * allowance on `reasoning_content` and returns zero `content`, which surfaces
+ * as the "思考上限" error.
  *
  *
- * Mirrors the protection already present in `buildAnthropicBodyWithReasoning`
- * (budget_tokens + 4096 answer reserve).
+ * This is a pure query. Two consumers act on it, both *before* the request
+ * body is built: `planLlmRequestBudget` raises the planned output to this
+ * floor (still bounded by the user's output cap and the context window), and
+ * the settings UI raises the user's configured output cap when they pick a
+ * reasoning level that needs more. Nothing may inflate `max_tokens` at
+ * body-build time — that happens after budgeting and would break the
+ * window conservation the planner just established.
  */
  */
-function thinkingMinMaxTokens(reasoning: ReasoningConfig): number {
+export function thinkingMinMaxTokens(reasoning: ReasoningConfig): number {
   switch (reasoning.mode) {
   switch (reasoning.mode) {
     case "low":
     case "low":
       return 4096
       return 4096
@@ -631,12 +639,24 @@ function thinkingMinMaxTokens(reasoning: ReasoningConfig): number {
   }
   }
 }
 }
 
 
-function ensureMinMaxTokens(body: Record<string, unknown>, min: number): void {
-  if (min <= 0) return
-  const current = body.max_tokens
-  if (typeof current !== "number" || current < min) {
-    body.max_tokens = min
-  }
+/**
+ * Whether explicit thinking can be honoured within the output allowance the
+ * caller already decided on. An absent `max_tokens` means the provider
+ * default applies and we have no basis to judge, so thinking stays on.
+ *
+ * OpenAI-compatible endpoints expose thinking as a boolean with no budget
+ * field, so the only remedy when it does not fit is to turn thinking off —
+ * unlike the Anthropic path, which can shrink `budget_tokens` instead.
+ */
+function thinkingFitsInOutputBudget(
+  body: Record<string, unknown>,
+  reasoning: ReasoningConfig,
+): boolean {
+  const required = thinkingMinMaxTokens(reasoning)
+  if (required <= 0) return true
+  const planned = body.max_tokens
+  if (typeof planned !== "number") return true
+  return planned >= required
 }
 }
 
 
 function isDeepSeekEndpoint(config: LlmConfig): boolean {
 function isDeepSeekEndpoint(config: LlmConfig): boolean {
@@ -644,22 +664,22 @@ function isDeepSeekEndpoint(config: LlmConfig): boolean {
 }
 }
 
 
 /**
 /**
- * Minimum context window for DeepSeek models. DeepSeek V3/V4 support
- * up to 1M tokens; the previous default of 64K/200K caused response
- * truncation on long inputs (bug report).
- */
-const DEEPSEEK_MIN_CONTEXT_SIZE = 1_000_000
-
-/**
- * Returns the effective maxContextSize for a given config, applying
- * model-specific minimums. DeepSeek endpoints get bumped to at least
- * 1M chars so long prompts aren't silently truncated.
+ * The context window to plan against, in tokens.
+ *
+ * Deliberately just the user's setting plus a fallback. Model-specific
+ * minimums used to be forced here, which meant the value in the settings UI
+ * and the value actually used could differ with nothing on screen to say so —
+ * for DeepSeek the window slider had no effect at all. Model defaults belong
+ * in the presets (`suggestedContextSize`), where the user can see and change
+ * them.
  */
  */
 export function getEffectiveMaxContextSize(config: LlmConfig): number {
 export function getEffectiveMaxContextSize(config: LlmConfig): number {
-  if (isDeepSeekEndpoint(config)) {
-    return Math.max(config.maxContextSize || 0, DEEPSEEK_MIN_CONTEXT_SIZE)
-  }
-  return config.maxContextSize || 204_800
+  return config.maxContextSize || MIN_USER_LLM_CONTEXT_SIZE
+}
+
+/** The declared output ceiling to plan against, in tokens. */
+export function getEffectiveMaxOutputTokens(config: LlmConfig): number {
+  return normalizeUserLlmMaxOutputTokens(config.maxOutputTokens)
 }
 }
 
 
 /**
 /**
@@ -781,8 +801,11 @@ function buildOpenAiCompatibleBody(
     if (reasoning.mode === "off") {
     if (reasoning.mode === "off") {
       body.thinking = { type: "disabled" }
       body.thinking = { type: "disabled" }
     } else if (reasoning.mode !== "auto") {
     } else if (reasoning.mode !== "auto") {
+      if (!thinkingFitsInOutputBudget(body, reasoning)) {
+        body.thinking = { type: "disabled" }
+        return body
+      }
       body.thinking = { type: "enabled" }
       body.thinking = { type: "enabled" }
-      ensureMinMaxTokens(body, thinkingMinMaxTokens(reasoning))
       const effort = reasoningEffort(reasoning)
       const effort = reasoningEffort(reasoning)
       if (effort) {
       if (effort) {
         body.reasoning_effort = effort
         body.reasoning_effort = effort
@@ -791,14 +814,18 @@ function buildOpenAiCompatibleBody(
     return body
     return body
   }
   }
 
 
+  // 思考放不下时改为关闭,同时压掉 reasoning_effort,避免请求体自相矛盾
+  let thinkingSuppressed = false
+
   // chat_template_kwargs 类型思考模型(Qwen3、MiMo)
   // chat_template_kwargs 类型思考模型(Qwen3、MiMo)
   // 同时检查模型名称和端点URL,双重保险确保MiMo等模型被正确识别
   // 同时检查模型名称和端点URL,双重保险确保MiMo等模型被正确识别
   if (isChatTemplateThinkingModel(config.model) || isMiMoEndpoint(config)) {
   if (isChatTemplateThinkingModel(config.model) || isMiMoEndpoint(config)) {
     if (reasoning.mode === "off") {
     if (reasoning.mode === "off") {
       body.chat_template_kwargs = { enable_thinking: false }
       body.chat_template_kwargs = { enable_thinking: false }
     } else if (reasoning.mode !== "auto") {
     } else if (reasoning.mode !== "auto") {
-      body.chat_template_kwargs = { enable_thinking: true }
-      ensureMinMaxTokens(body, thinkingMinMaxTokens(reasoning))
+      const fits = thinkingFitsInOutputBudget(body, reasoning)
+      body.chat_template_kwargs = { enable_thinking: fits }
+      if (!fits) thinkingSuppressed = true
     }
     }
   }
   }
 
 
@@ -807,13 +834,18 @@ function buildOpenAiCompatibleBody(
     if (reasoning.mode === "off") {
     if (reasoning.mode === "off") {
       body.thinking = { type: "disabled" }
       body.thinking = { type: "disabled" }
     } else if (reasoning.mode !== "auto") {
     } else if (reasoning.mode !== "auto") {
-      body.thinking = { type: "enabled" }
-      ensureMinMaxTokens(body, thinkingMinMaxTokens(reasoning))
+      const fits = thinkingFitsInOutputBudget(body, reasoning)
+      body.thinking = { type: fits ? "enabled" : "disabled" }
+      if (!fits) thinkingSuppressed = true
     }
     }
   }
   }
 
 
   const effort = reasoningEffort(reasoning)
   const effort = reasoningEffort(reasoning)
-  if ((config.provider === "openai" || config.provider === "azure" || config.provider === "custom") && effort) {
+  if (
+    !thinkingSuppressed
+    && (config.provider === "openai" || config.provider === "azure" || config.provider === "custom")
+    && effort
+  ) {
     body.reasoning_effort = effort
     body.reasoning_effort = effort
   }
   }
 
 
@@ -936,12 +968,13 @@ function buildAnthropicBodyWithReasoning(
         : reasoning.mode === "medium"
         : reasoning.mode === "medium"
           ? 4096
           ? 4096
         : 8192
         : 8192
-  const budgetTokens = Math.max(1024, budget)
-  const minAnswerTokens = 4096
-  const minTotalTokens = budgetTokens + minAnswerTokens
-  if ((body.max_tokens as number) < minTotalTokens) {
-    body.max_tokens = minTotalTokens
-  }
+  const maxOutputTokens = body.max_tokens as number
+  const minimumAnswerTokens = 512
+  const availableThinkingTokens = maxOutputTokens - minimumAnswerTokens
+  // Anthropic requires at least 1024 thinking tokens. Never inflate max_tokens
+  // beyond the workflow plan; disable explicit thinking when it cannot fit.
+  if (availableThinkingTokens < 1024) return body
+  const budgetTokens = Math.min(Math.max(1024, budget), availableThinkingTokens)
   body.thinking = { type: "enabled", budget_tokens: budgetTokens }
   body.thinking = { type: "enabled", budget_tokens: budgetTokens }
   delete body.temperature
   delete body.temperature
   delete body.top_p
   delete body.top_p

+ 50 - 9
src/lib/novel/context-engine.spec.ts

@@ -1,4 +1,6 @@
-import { describe, expect, it } from "vitest"
+import { afterEach, beforeEach, describe, expect, it } from "vitest"
+import i18n from "@/i18n"
+import { charsPerTokenForLanguage } from "@/lib/context-budget"
 import { annotateChapterOutlineStatus, contextPackToPrompt, trimContextPack, type ContextPack } from "./context-engine"
 import { annotateChapterOutlineStatus, contextPackToPrompt, trimContextPack, type ContextPack } from "./context-engine"
 
 
 const basePack: ContextPack = {
 const basePack: ContextPack = {
@@ -56,6 +58,16 @@ describe("annotateChapterOutlineStatus", () => {
 })
 })
 
 
 describe("trimContextPack 两级裁剪", () => {
 describe("trimContextPack 两级裁剪", () => {
+  // Trim-level cases were authored against the old hardcoded ×4 char quota.
+  // Pin English density so tokenBudget→chars stays comparable; CJK coverage
+  // lives in the language-density nested suite below.
+  beforeEach(async () => {
+    await i18n.changeLanguage("en")
+  })
+  afterEach(async () => {
+    await i18n.changeLanguage("zh")
+  })
+
   const fullPack: ContextPack = {
   const fullPack: ContextPack = {
     task: "生成第10章正文",
     task: "生成第10章正文",
     chapterGoal: "主角遭遇反派,爆发冲突",
     chapterGoal: "主角遭遇反派,爆发冲突",
@@ -111,19 +123,16 @@ describe("trimContextPack 两级裁剪", () => {
   })
   })
 
 
   it("第二级裁剪:放不下的字段做内容精简", () => {
   it("第二级裁剪:放不下的字段做内容精简", () => {
+    // Minimal pack so the only overflowing field is relatedSettings — forces
+    // the partial-trim path instead of whole-field drops.
     const smallPack: ContextPack = {
     const smallPack: ContextPack = {
-      ...fullPack,
-      searchResults: "",
-      graphSearchResults: "",
-      nextChapterAdvice: "",
+      ...basePack,
+      task: "测试",
       relatedSettings: "世界设定内容。" + "详细描述。".repeat(100),
       relatedSettings: "世界设定内容。" + "详细描述。".repeat(100),
-      canonRules: "",
-      writingStyle: "",
-      timeline: "",
-      revisionDirectives: "",
     }
     }
     const result = trimContextPack(smallPack, 100)
     const result = trimContextPack(smallPack, 100)
     expect(result.partiallyTrimmedField).toBeDefined()
     expect(result.partiallyTrimmedField).toBeDefined()
+    expect(result.partiallyTrimmedField?.fieldKey).toBe("relatedSettings")
     expect(result.partiallyTrimmedField?.keptChars).toBeLessThan(result.partiallyTrimmedField?.originalChars ?? 0)
     expect(result.partiallyTrimmedField?.keptChars).toBeLessThan(result.partiallyTrimmedField?.originalChars ?? 0)
   })
   })
 
 
@@ -172,4 +181,36 @@ describe("trimContextPack 两级裁剪", () => {
     expect(result.prompt).not.toContain("第3章:试炼")
     expect(result.prompt).not.toContain("第3章:试炼")
     expect(result.prompt).toContain("生成第10章正文")
     expect(result.prompt).toContain("生成第10章正文")
   })
   })
+
+  describe("语言密度参与字符配额", () => {
+    afterEach(async () => {
+      await i18n.changeLanguage("zh")
+    })
+
+    it("同 token 预算下 CJK 字符配额约为英文的 1/4,并更早触发裁剪", async () => {
+      const tokenBudget = 500
+      // Fits English (×4 → 2000 chars) but not CJK (×1 → 500 chars).
+      const pack: ContextPack = {
+        ...basePack,
+        task: "测试语言密度",
+        soulDoc: "文".repeat(1_200),
+      }
+
+      expect(charsPerTokenForLanguage("zh")).toBe(charsPerTokenForLanguage("en") / 4)
+
+      await i18n.changeLanguage("zh")
+      expect(charsPerTokenForLanguage()).toBe(1)
+      const cjk = trimContextPack(pack, tokenBudget)
+      expect(
+        cjk.trimmedFields.length + (cjk.partiallyTrimmedField ? 1 : 0),
+      ).toBeGreaterThan(0)
+
+      await i18n.changeLanguage("en")
+      expect(charsPerTokenForLanguage()).toBe(4)
+      const en = trimContextPack(pack, tokenBudget)
+      expect(en.trimmedFields).toHaveLength(0)
+      expect(en.partiallyTrimmedField).toBeUndefined()
+      expect(en.finalChars).toBeGreaterThan(cjk.finalChars)
+    })
+  })
 })
 })

+ 7 - 2
src/lib/novel/context-engine.ts

@@ -1,4 +1,7 @@
-import { resolveContextPackTokenBudget } from "@/lib/context-budget"
+import {
+  charsPerTokenForLanguage,
+  resolveContextPackTokenBudget,
+} from "@/lib/context-budget"
 import { listDirectory, readFile } from "@/commands/fs"
 import { listDirectory, readFile } from "@/commands/fs"
 import i18n from "@/i18n"
 import i18n from "@/i18n"
 import { searchWiki, tokenizeQuery } from "@/lib/search"
 import { searchWiki, tokenizeQuery } from "@/lib/search"
@@ -1198,7 +1201,9 @@ export function trimContextPack(
   const resolvedTokenBudget = tokenBudget && tokenBudget > 0
   const resolvedTokenBudget = tokenBudget && tokenBudget > 0
     ? tokenBudget
     ? tokenBudget
     : resolveContextPackTokenBudget({ maxContextSize: options?.maxContextSize })
     : resolveContextPackTokenBudget({ maxContextSize: options?.maxContextSize })
-  const targetChars = resolvedTokenBudget * 4
+  // Match the token estimator: CJK ≈ 1 char/token, English ≈ 4 chars/token.
+  // A hardcoded ×4 over-admits Chinese after maxContextSize became real tokens.
+  const targetChars = Math.floor(resolvedTokenBudget * charsPerTokenForLanguage())
 
 
   if (totalChars <= targetChars) {
   if (totalChars <= targetChars) {
     for (const { title, content } of fieldData) {
     for (const { title, content } of fieldData) {

+ 44 - 3
src/lib/novel/deep-chapter-generation.spec.ts

@@ -1211,7 +1211,7 @@ describe("runDeepChapterGeneration", () => {
     expect(capturedPrompts[1]).toContain("冷峻克制")
     expect(capturedPrompts[1]).toContain("冷峻克制")
   })
   })
 
 
-  it("does not pass app-side max_tokens limits to deep chapter model calls", async () => {
+  it("binds analysis and generation budgets to deep chapter model calls", async () => {
     const deps = createDeps()
     const deps = createDeps()
     const overrides: Array<RequestOverrides | undefined> = []
     const overrides: Array<RequestOverrides | undefined> = []
     vi.mocked(deps.streamChat).mockImplementation(async (
     vi.mocked(deps.streamChat).mockImplementation(async (
@@ -1235,13 +1235,49 @@ describe("runDeepChapterGeneration", () => {
     })
     })
 
 
     await runDeepChapterGeneration(
     await runDeepChapterGeneration(
+      {
+        projectPath: "E:/Novel",
+        userRequest: "生成第三章",
+        chapterNumber: 3,
+        // Auto reasoning carries no output floor, so the per-stage budgets show through.
+        llmConfig: { ...llmConfig, reasoning: { mode: "auto" } },
+      },
+      {},
+      deps,
+    )
+
+    expect(overrides.length).toBeGreaterThan(0)
+    expect(overrides.every((item) => typeof item?.max_tokens === "number")).toBe(true)
+    expect(overrides.some((item) => item?.max_tokens === 4_096)).toBe(true)
+    expect(overrides.some((item) => item?.max_tokens === 8_000)).toBe(true)
+  })
+
+  it("raises stage output to the floor the configured reasoning level needs", async () => {
+    const deps = createDeps()
+    const overrides: Array<RequestOverrides | undefined> = []
+    vi.mocked(deps.streamChat).mockImplementation(async (
+      _config: LlmConfig,
+      messages: ChatMessage[],
+      callbacks: StreamCallbacks,
+      _signal,
+      requestOverrides,
+    ) => {
+      overrides.push(requestOverrides)
+      const prompt = messagesPromptText(messages)
+      callbacks.onToken(prompt.includes("正文") ? chapterText("思考档位正文", 3000) : "写作任务书内容")
+      callbacks.onDone()
+    })
+
+    await runDeepChapterGeneration(
+      // llmConfig requests "high" reasoning, which needs 16384 output tokens
+      // before it can emit any final content.
       { projectPath: "E:/Novel", userRequest: "生成第三章", chapterNumber: 3, llmConfig },
       { projectPath: "E:/Novel", userRequest: "生成第三章", chapterNumber: 3, llmConfig },
       {},
       {},
       deps,
       deps,
     )
     )
 
 
     expect(overrides.length).toBeGreaterThan(0)
     expect(overrides.length).toBeGreaterThan(0)
-    expect(overrides.every((item) => item?.max_tokens === undefined)).toBe(true)
+    expect(overrides.every((item) => item?.max_tokens === 16_384)).toBe(true)
   })
   })
 
 
   it("preserves configured model reasoning for chapter generation calls", async () => {
   it("preserves configured model reasoning for chapter generation calls", async () => {
@@ -1309,7 +1345,12 @@ describe("runDeepChapterGeneration", () => {
 
 
     expect(result.finalContent).toContain("最终兜底正文")
     expect(result.finalContent).toContain("最终兜底正文")
     expect(overrides[0]?.reasoning).toBeUndefined()
     expect(overrides[0]?.reasoning).toBeUndefined()
-    expect(overrides[1]).toEqual({ reasoning: { mode: "off" } })
+    expect(overrides[1]).toEqual({
+      // Budgets are planned once per run, from the configured "high" reasoning
+      // level; the retry only turns thinking off, and max_tokens is a ceiling.
+      max_tokens: 16_384,
+      reasoning: { mode: "off" },
+    })
   })
   })
 
 
   it("uses fast, standard, and strict workflow routes", async () => {
   it("uses fast, standard, and strict workflow routes", async () => {

+ 42 - 14
src/lib/novel/deep-chapter-generation.ts

@@ -13,7 +13,14 @@ import {
   isReasoningOnlyResponseError,
   isReasoningOnlyResponseError,
   withReasoningDisabled,
   withReasoningDisabled,
 } from "@/lib/reasoning-retry";
 } from "@/lib/reasoning-retry";
-import { computeWritingContextPackTokenBudget } from "@/lib/context-budget";
+import {
+  charsPerTokenForLanguage,
+  planChapterRequestBudget,
+} from "@/lib/context-budget";
+import {
+  getEffectiveMaxOutputTokens,
+  thinkingMinMaxTokens,
+} from "@/lib/llm-providers";
 import { USER_ABORT_MESSAGE, rethrowIfUserAbort, throwIfAborted } from "@/lib/user-abort";
 import { USER_ABORT_MESSAGE, rethrowIfUserAbort, throwIfAborted } from "@/lib/user-abort";
 import {
 import {
   buildContextPack,
   buildContextPack,
@@ -146,10 +153,6 @@ const defaultDeps: DeepChapterGenerationDeps = {
 const REPEAT_CHECK_MIN_CHARS = 600;
 const REPEAT_CHECK_MIN_CHARS = 600;
 const REPEAT_WINDOW_CHARS = 120;
 const REPEAT_WINDOW_CHARS = 120;
 const REPEAT_HIT_LIMIT = 3;
 const REPEAT_HIT_LIMIT = 3;
-/** chars/token approximation used to convert the token budget to characters
- *  for the outline cap (mirrors context-budget.ts / contextPackToPrompt). */
-const DEEP_CHAPTER_CHARS_PER_TOKEN = 4;
-
 function hasUsableChapterExecutionContract(contract: ChapterExecutionContract | null): contract is ChapterExecutionContract {
 function hasUsableChapterExecutionContract(contract: ChapterExecutionContract | null): contract is ChapterExecutionContract {
   if (!contract) return false;
   if (!contract) return false;
   const hasSceneChecks = contract.sceneSteps.some((step) =>
   const hasSceneChecks = contract.sceneSteps.some((step) =>
@@ -667,13 +670,36 @@ export async function runDeepChapterGeneration(
     : input.llmConfig.maxContextSize;
     : input.llmConfig.maxContextSize;
 
 
   // 大纲与其余上下文共用同一窗口预算:按单章目标字数×2预留输出,再分配资料包。
   // 大纲与其余上下文共用同一窗口预算:按单章目标字数×2预留输出,再分配资料包。
-  const totalContextTokenBudget = computeWritingContextPackTokenBudget({
+  const sharedMaxOutputTokens = getEffectiveMaxOutputTokens(input.llmConfig);
+  const sharedThinkingFloor = thinkingMinMaxTokens(
+    input.llmConfig.reasoning ?? { mode: "auto" },
+  );
+  const chapterAnalysisBudget = planChapterRequestBudget({
     maxContextSize: sharedContextWindow,
     maxContextSize: sharedContextWindow,
     contextTokenBudget: novelConfig.contextTokenBudget,
     contextTokenBudget: novelConfig.contextTokenBudget,
     chapterTargetChars: novelConfig.chapterTargetChars,
     chapterTargetChars: novelConfig.chapterTargetChars,
+    stage: "analysis",
+    maxOutputTokens: sharedMaxOutputTokens,
+    thinkingFloorTokens: sharedThinkingFloor,
   });
   });
-  const totalContextCharBudget =
-    totalContextTokenBudget * DEEP_CHAPTER_CHARS_PER_TOKEN;
+  const chapterGenerationBudget = planChapterRequestBudget({
+    maxContextSize: sharedContextWindow,
+    contextTokenBudget: novelConfig.contextTokenBudget,
+    chapterTargetChars: novelConfig.chapterTargetChars,
+    stage: "generation",
+    maxOutputTokens: sharedMaxOutputTokens,
+    thinkingFloorTokens: sharedThinkingFloor,
+  });
+  const totalContextTokenBudget = chapterGenerationBudget.contextTokenBudget;
+  const analysisRequestOverrides: RequestOverrides = {
+    max_tokens: chapterAnalysisBudget.outputTokens,
+  };
+  const generationRequestOverrides: RequestOverrides = {
+    max_tokens: chapterGenerationBudget.outputTokens,
+  };
+  // Same density as the token estimator / trimContextPack (CJK 1, English 4).
+  const charsPerToken = charsPerTokenForLanguage();
+  const totalContextCharBudget = totalContextTokenBudget * charsPerToken;
   const outlineCharCap = Math.floor(
   const outlineCharCap = Math.floor(
     totalContextCharBudget * DEEP_CHAPTER_OUTLINE_MAX_FRAC,
     totalContextCharBudget * DEEP_CHAPTER_OUTLINE_MAX_FRAC,
   );
   );
@@ -705,7 +731,7 @@ export async function runDeepChapterGeneration(
   const restContextTokenBudget = Math.max(
   const restContextTokenBudget = Math.max(
     DEEP_CHAPTER_REST_TOKEN_FLOOR,
     DEEP_CHAPTER_REST_TOKEN_FLOOR,
     totalContextTokenBudget -
     totalContextTokenBudget -
-      Math.ceil(outlineText.length / DEEP_CHAPTER_CHARS_PER_TOKEN),
+      Math.ceil(outlineText.length / charsPerToken),
   );
   );
   const contextPrompt = [
   const contextPrompt = [
     previousChaptersAnalysis
     previousChaptersAnalysis
@@ -793,7 +819,7 @@ export async function runDeepChapterGeneration(
             callbacks.onThinking?.(
             callbacks.onThinking?.(
               formatStageThinking("阶段2:写作任务书", partial),
               formatStageThinking("阶段2:写作任务书", partial),
             ),
             ),
-          undefined,
+          analysisRequestOverrides,
           cachePrefix,
           cachePrefix,
         ),
         ),
       (value) => `写作任务书完成,约 ${countChapterChars(value)} 字。`,
       (value) => `写作任务书完成,约 ${countChapterChars(value)} 字。`,
@@ -867,7 +893,7 @@ export async function runDeepChapterGeneration(
             callbacks.onThinking?.(
             callbacks.onThinking?.(
               formatStageThinking("阶段3:正文初稿", partial),
               formatStageThinking("阶段3:正文初稿", partial),
             ),
             ),
-          undefined,
+          generationRequestOverrides,
           cachePrefix,
           cachePrefix,
         ),
         ),
       (value) => `正文初稿完成,约 ${countChapterChars(value)} 字。`,
       (value) => `正文初稿完成,约 ${countChapterChars(value)} 字。`,
@@ -907,7 +933,7 @@ export async function runDeepChapterGeneration(
               callbacks.onThinking?.(
               callbacks.onThinking?.(
                 formatStageThinking("阶段3:正文扩写补足", partial),
                 formatStageThinking("阶段3:正文扩写补足", partial),
               ),
               ),
-            undefined,
+            generationRequestOverrides,
             cachePrefix,
             cachePrefix,
           ),
           ),
         (value) => `正文扩写补足完成,约 ${countChapterChars(value)} 字。`,
         (value) => `正文扩写补足完成,约 ${countChapterChars(value)} 字。`,
@@ -1205,7 +1231,7 @@ export async function runDeepChapterGeneration(
             callbacks.onThinking?.(
             callbacks.onThinking?.(
               formatStageThinking("阶段5:自动返修", partial),
               formatStageThinking("阶段5:自动返修", partial),
             ),
             ),
-          undefined,
+          generationRequestOverrides,
           cachePrefix,
           cachePrefix,
         ),
         ),
       (value) =>
       (value) =>
@@ -1358,6 +1384,7 @@ export async function runDeepChapterGeneration(
             signal,
             signal,
             customDeAiSkill || undefined,
             customDeAiSkill || undefined,
             cachePrefix,
             cachePrefix,
+            generationRequestOverrides,
           ),
           ),
         (value) =>
         (value) =>
           `简单审查与去AI味完成,最终正文约 ${countChapterChars(value)} 字。`,
           `简单审查与去AI味完成,最终正文约 ${countChapterChars(value)} 字。`,
@@ -1706,6 +1733,7 @@ async function finalPolishChapter(
   signal?: AbortSignal,
   signal?: AbortSignal,
   customDeAiSkill?: string,
   customDeAiSkill?: string,
   cachePrefix?: string,
   cachePrefix?: string,
+  requestOverrides?: RequestOverrides,
 ): Promise<string> {
 ): Promise<string> {
   assertNotAborted(signal);
   assertNotAborted(signal);
   callbacks.onThinking?.(
   callbacks.onThinking?.(
@@ -1737,7 +1765,7 @@ async function finalPolishChapter(
       callbacks.onThinking?.(
       callbacks.onThinking?.(
         formatStageThinking("阶段6:简单审查与去AI味", partial),
         formatStageThinking("阶段6:简单审查与去AI味", partial),
       ),
       ),
-    undefined,
+    requestOverrides,
     cachePrefix,
     cachePrefix,
   );
   );
   assertNotAborted(signal);
   assertNotAborted(signal);

+ 11 - 5
src/lib/novel/model-resolver.ts

@@ -1,9 +1,10 @@
 import { useWikiStore, type LlmConfig, type NovelConfig, type ProviderOverride } from "@/stores/wiki-store"
 import { useWikiStore, type LlmConfig, type NovelConfig, type ProviderOverride } from "@/stores/wiki-store"
-import { LLM_PRESETS } from "@/components/settings/llm-presets"
+import { findLlmPresetById } from "@/components/settings/llm-presets"
 import { resolveConfig } from "@/components/settings/preset-resolver"
 import { resolveConfig } from "@/components/settings/preset-resolver"
 import { hasUsableLlm } from "@/lib/has-usable-llm"
 import { hasUsableLlm } from "@/lib/has-usable-llm"
-import { getEffectiveMaxContextSize } from "@/lib/llm-providers"
+import { getEffectiveMaxContextSize, getEffectiveMaxOutputTokens } from "@/lib/llm-providers"
 import { getStableAvailableModelKey, getEffectiveSavedModels } from "@/lib/llm-model-keys"
 import { getStableAvailableModelKey, getEffectiveSavedModels } from "@/lib/llm-model-keys"
+import { normalizeUserLlmConfig } from "@/lib/llm-context-size"
 
 
 export type NovelTaskType = "writing" | "review" | "summary" | "extract" | "lint" | "deAi"
 export type NovelTaskType = "writing" | "review" | "summary" | "extract" | "lint" | "deAi"
 
 
@@ -18,7 +19,12 @@ function isConfigUsable(cfg: LlmConfig, providerConfigs: Record<string, Provider
 }
 }
 
 
 function withEffectiveContextSize(config: LlmConfig): LlmConfig {
 function withEffectiveContextSize(config: LlmConfig): LlmConfig {
-  return { ...config, maxContextSize: getEffectiveMaxContextSize(config) }
+  const normalized = normalizeUserLlmConfig(config)
+  return {
+    ...normalized,
+    maxContextSize: getEffectiveMaxContextSize(normalized),
+    maxOutputTokens: getEffectiveMaxOutputTokens(normalized),
+  }
 }
 }
 
 
 function toUnusableConfig(baseConfig: LlmConfig): LlmConfig {
 function toUnusableConfig(baseConfig: LlmConfig): LlmConfig {
@@ -72,7 +78,7 @@ export function resolveModelConfig(
     const modelId = targetModel.slice(slashIdx + 1)
     const modelId = targetModel.slice(slashIdx + 1)
     const override = providerConfigs[providerId]
     const override = providerConfigs[providerId]
     if (override && getEffectiveSavedModels(override).some((m) => m.model === modelId)) {
     if (override && getEffectiveSavedModels(override).some((m) => m.model === modelId)) {
-      const template = LLM_PRESETS.find((p) => p.id === providerId) ?? LLM_PRESETS.find((p) => p.id === "custom")
+      const template = findLlmPresetById(providerId) ?? findLlmPresetById("custom")
       if (template) {
       if (template) {
         return withEffectiveContextSize({ ...resolveConfig(template, override, baseConfig), model: modelId })
         return withEffectiveContextSize({ ...resolveConfig(template, override, baseConfig), model: modelId })
       }
       }
@@ -82,7 +88,7 @@ export function resolveModelConfig(
   // 回退:按纯模型名匹配(兼容旧数据)
   // 回退:按纯模型名匹配(兼容旧数据)
   for (const [providerId, override] of Object.entries(providerConfigs)) {
   for (const [providerId, override] of Object.entries(providerConfigs)) {
     if (getEffectiveSavedModels(override).some((m) => m.model === targetModel)) {
     if (getEffectiveSavedModels(override).some((m) => m.model === targetModel)) {
-      const template = LLM_PRESETS.find((p) => p.id === providerId) ?? LLM_PRESETS.find((p) => p.id === "custom")
+      const template = findLlmPresetById(providerId) ?? findLlmPresetById("custom")
       if (template) {
       if (template) {
         return withEffectiveContextSize({ ...resolveConfig(template, override, baseConfig), model: targetModel })
         return withEffectiveContextSize({ ...resolveConfig(template, override, baseConfig), model: targetModel })
       }
       }

+ 76 - 1
src/lib/project-store.integration.test.ts

@@ -7,7 +7,7 @@
  */
  */
 import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"
 import { describe, it, expect, beforeEach, afterEach, vi } from "vitest"
 import { realFs, createTempProject, writeFileRaw, fileExists, readFileRaw } from "@/test-helpers/fs-temp"
 import { realFs, createTempProject, writeFileRaw, fileExists, readFileRaw } from "@/test-helpers/fs-temp"
-import type { NovelConfig, RevisionFeedbackWindowConfig, SourceWatchConfig, RerankConfig } from "@/stores/wiki-store"
+import type { LlmConfig, NovelConfig, ProviderConfigs, RevisionFeedbackWindowConfig, SourceWatchConfig, RerankConfig } from "@/stores/wiki-store"
 import type { McpConfig } from "@/lib/mcp/config"
 import type { McpConfig } from "@/lib/mcp/config"
 
 
 vi.mock("@/commands/fs", () => realFs)
 vi.mock("@/commands/fs", () => realFs)
@@ -38,6 +38,8 @@ import {
   loadSourceWatchConfig,
   loadSourceWatchConfig,
   saveMcpConfig,
   saveMcpConfig,
   loadMcpConfig,
   loadMcpConfig,
+  loadLlmConfig,
+  loadProviderConfigs,
 } from "./project-store"
 } from "./project-store"
 
 
 let tmp: { path: string; cleanup: () => Promise<void> }
 let tmp: { path: string; cleanup: () => Promise<void> }
@@ -51,6 +53,79 @@ afterEach(async () => {
   await tmp.cleanup()
   await tmp.cleanup()
 })
 })
 
 
+it("silently migrates legacy main and provider context sizes on load", async () => {
+  const llmConfig: LlmConfig = {
+    provider: "custom",
+    apiKey: "key",
+    model: "model-a",
+    customEndpoint: "https://example.com/v1",
+    ollamaUrl: "",
+    maxContextSize: 128_000,
+  }
+  const providerConfigs: ProviderConfigs = {
+    custom: {
+      apiKey: "key",
+      model: "model-a",
+      maxContextSize: 32_768,
+      enabled: false,
+    },
+  }
+  inMemoryStore.set("llmConfig", llmConfig)
+  inMemoryStore.set("providerConfigs", providerConfigs)
+
+  expect((await loadLlmConfig())?.maxContextSize).toBe(204_800)
+  expect((await loadProviderConfigs())?.custom?.maxContextSize).toBe(204_800)
+  expect((inMemoryStore.get("llmConfig") as LlmConfig).maxContextSize).toBe(204_800)
+  expect((inMemoryStore.get("providerConfigs") as ProviderConfigs).custom?.enabled).toBe(false)
+})
+
+describe("DeepSeek window migration", () => {
+  it("lifts a stale saved window to 1M once, then leaves the user in control", async () => {
+    // The runtime used to force DeepSeek to 1M, hiding whatever was saved.
+    // Removing that forcing would expose these stale values, so they are
+    // lifted once — after which a deliberate reduction must stick.
+    inMemoryStore.set("llmConfig", {
+      provider: "custom",
+      apiKey: "key",
+      model: "deepseek-chat",
+      customEndpoint: "https://api.deepseek.com/v1",
+      ollamaUrl: "",
+      maxContextSize: 262_144,
+    } satisfies LlmConfig)
+    inMemoryStore.set("providerConfigs", {
+      deepseek: { apiKey: "key", model: "deepseek-chat", maxContextSize: 262_144 },
+    } satisfies ProviderConfigs)
+
+    expect((await loadLlmConfig())?.maxContextSize).toBe(1_000_000)
+    expect((await loadProviderConfigs())?.deepseek?.maxContextSize).toBe(1_000_000)
+
+    inMemoryStore.set("llmConfig", {
+      ...(inMemoryStore.get("llmConfig") as LlmConfig),
+      maxContextSize: 262_144,
+    })
+    inMemoryStore.set("providerConfigs", {
+      deepseek: { apiKey: "key", model: "deepseek-chat", maxContextSize: 262_144 },
+    } satisfies ProviderConfigs)
+
+    expect((await loadLlmConfig())?.maxContextSize).toBe(262_144)
+    expect((await loadProviderConfigs())?.deepseek?.maxContextSize).toBe(262_144)
+  })
+
+  it("leaves third-party hosts serving DeepSeek models alone", async () => {
+    // 1M is DeepSeek's own figure; gateways reselling the model set their own.
+    inMemoryStore.set("llmConfig", {
+      provider: "custom",
+      apiKey: "key",
+      model: "deepseek-ai/deepseek-v4-pro",
+      customEndpoint: "https://api.atlascloud.ai/v1",
+      ollamaUrl: "",
+      maxContextSize: 262_144,
+    } satisfies LlmConfig)
+
+    expect((await loadLlmConfig())?.maxContextSize).toBe(262_144)
+  })
+})
+
 function makeNovelConfig(overrides: Partial<NovelConfig> = {}): NovelConfig {
 function makeNovelConfig(overrides: Partial<NovelConfig> = {}): NovelConfig {
   return {
   return {
     contextTokenBudget: 200000,
     contextTokenBudget: 200000,

+ 82 - 4
src/lib/project-store.ts

@@ -14,6 +14,10 @@ import {
 } from "@/lib/visual-style-settings"
 } from "@/lib/visual-style-settings"
 import { normalizePath } from "@/lib/path-utils"
 import { normalizePath } from "@/lib/path-utils"
 import { readFile, writeFile, fileExists } from "@/commands/fs"
 import { readFile, writeFile, fileExists } from "@/commands/fs"
+import {
+  normalizeProviderConfigs,
+  normalizeUserLlmConfig,
+} from "@/lib/llm-context-size"
 
 
 const RECENT_PROJECTS_KEY = "recentProjects"
 const RECENT_PROJECTS_KEY = "recentProjects"
 const LAST_PROJECT_KEY = "lastProject"
 const LAST_PROJECT_KEY = "lastProject"
@@ -47,6 +51,49 @@ export async function addToRecentProjects(
 }
 }
 
 
 const LLM_CONFIG_KEY = "llmConfig"
 const LLM_CONFIG_KEY = "llmConfig"
+// Separate markers per store slot: the two loaders run independently and in no
+// guaranteed order, so a shared marker would let whichever ran first cancel the
+// other's migration.
+const DEEPSEEK_WINDOW_MIGRATION_KEYS = {
+  llmConfig: "deepseekWindowMigratedV1.llmConfig",
+  providerConfigs: "deepseekWindowMigratedV1.providerConfigs",
+} as const
+/** DeepSeek's official published context window. */
+const DEEPSEEK_OFFICIAL_CONTEXT_SIZE = 1_000_000
+/** Preset id whose configuration is known to target api.deepseek.com. */
+const DEEPSEEK_PRESET_ID = "deepseek"
+
+function isDeepSeekOfficialEndpoint(endpoint: string | undefined): boolean {
+  return typeof endpoint === "string" && /api\.deepseek\.com/i.test(endpoint)
+}
+
+/**
+ * One-time lift of saved DeepSeek windows to the official 1M.
+ *
+ * The window used to be forced to 1M at request time, which hid whatever the
+ * user had actually saved. Now that the forcing is gone those stale values
+ * would take effect, so they get raised once — in the user's own settings,
+ * where they can see and change it. The marker makes this genuinely one-time:
+ * without it, anyone who deliberately lowered the window afterwards would find
+ * it raised again on every launch, which is the hardcoding we just removed.
+ *
+ * Scoped to DeepSeek's own endpoint. Third-party hosts serving DeepSeek models
+ * (Atlas Cloud, Ollama Cloud, Volcengine) set their own limits, and the 1M
+ * figure has no authority there.
+ */
+async function hasRunDeepSeekWindowMigration(
+  slot: keyof typeof DEEPSEEK_WINDOW_MIGRATION_KEYS,
+): Promise<boolean> {
+  const store = await getStore()
+  return (await store.get<boolean>(DEEPSEEK_WINDOW_MIGRATION_KEYS[slot])) === true
+}
+
+async function markDeepSeekWindowMigrationDone(
+  slot: keyof typeof DEEPSEEK_WINDOW_MIGRATION_KEYS,
+): Promise<void> {
+  const store = await getStore()
+  await store.set(DEEPSEEK_WINDOW_MIGRATION_KEYS[slot], true)
+}
 const AI_CHAT_MODEL_KEY = "aiChatModel"
 const AI_CHAT_MODEL_KEY = "aiChatModel"
 const AI_OUTLINE_MODEL_KEY = "aiOutlineModel"
 const AI_OUTLINE_MODEL_KEY = "aiOutlineModel"
 let aiOutlineModelSaveRevision = 0
 let aiOutlineModelSaveRevision = 0
@@ -57,12 +104,25 @@ const ACTIVE_PRESET_KEY = "activePresetId"
 
 
 export async function saveLlmConfig(config: LlmConfig): Promise<void> {
 export async function saveLlmConfig(config: LlmConfig): Promise<void> {
   const store = await getStore()
   const store = await getStore()
-  await store.set(LLM_CONFIG_KEY, config)
+  await store.set(LLM_CONFIG_KEY, normalizeUserLlmConfig(config))
 }
 }
 
 
 export async function loadLlmConfig(): Promise<LlmConfig | null> {
 export async function loadLlmConfig(): Promise<LlmConfig | null> {
   const store = await getStore()
   const store = await getStore()
-  return (await store.get<LlmConfig>(LLM_CONFIG_KEY)) ?? null
+  const saved = (await store.get<LlmConfig>(LLM_CONFIG_KEY)) ?? null
+  if (!saved) return null
+  let normalized = normalizeUserLlmConfig(saved)
+  if (!(await hasRunDeepSeekWindowMigration("llmConfig"))) {
+    if (
+      isDeepSeekOfficialEndpoint(normalized.customEndpoint)
+      && normalized.maxContextSize < DEEPSEEK_OFFICIAL_CONTEXT_SIZE
+    ) {
+      normalized = { ...normalized, maxContextSize: DEEPSEEK_OFFICIAL_CONTEXT_SIZE }
+    }
+    await markDeepSeekWindowMigrationDone("llmConfig")
+  }
+  if (normalized !== saved) await store.set(LLM_CONFIG_KEY, normalized)
+  return normalized
 }
 }
 
 
 export async function saveAiChatModel(model: string): Promise<void> {
 export async function saveAiChatModel(model: string): Promise<void> {
@@ -105,12 +165,30 @@ export async function loadDefaultLlmModel(): Promise<string | null> {
 
 
 export async function saveProviderConfigs(configs: ProviderConfigs): Promise<void> {
 export async function saveProviderConfigs(configs: ProviderConfigs): Promise<void> {
   const store = await getStore()
   const store = await getStore()
-  await store.set(PROVIDER_CONFIGS_KEY, configs)
+  await store.set(PROVIDER_CONFIGS_KEY, normalizeProviderConfigs(configs))
 }
 }
 
 
 export async function loadProviderConfigs(): Promise<ProviderConfigs | null> {
 export async function loadProviderConfigs(): Promise<ProviderConfigs | null> {
   const store = await getStore()
   const store = await getStore()
-  return (await store.get<ProviderConfigs>(PROVIDER_CONFIGS_KEY)) ?? null
+  const saved = (await store.get<ProviderConfigs>(PROVIDER_CONFIGS_KEY)) ?? null
+  if (!saved) return null
+  let normalized = normalizeProviderConfigs(saved)
+  if (!(await hasRunDeepSeekWindowMigration("providerConfigs"))) {
+    const deepseek = normalized[DEEPSEEK_PRESET_ID]
+    if (
+      deepseek
+      && deepseek.maxContextSize !== undefined
+      && deepseek.maxContextSize < DEEPSEEK_OFFICIAL_CONTEXT_SIZE
+    ) {
+      normalized = {
+        ...normalized,
+        [DEEPSEEK_PRESET_ID]: { ...deepseek, maxContextSize: DEEPSEEK_OFFICIAL_CONTEXT_SIZE },
+      }
+    }
+    await markDeepSeekWindowMigrationDone("providerConfigs")
+  }
+  if (normalized !== saved) await store.set(PROVIDER_CONFIGS_KEY, normalized)
+  return normalized
 }
 }
 
 
 export async function saveActivePresetId(id: string | null): Promise<void> {
 export async function saveActivePresetId(id: string | null): Promise<void> {

+ 14 - 3
src/stores/wiki-store.ts

@@ -31,6 +31,10 @@ import {
 } from "@/lib/visual-style-settings"
 } from "@/lib/visual-style-settings"
 import { DEFAULT_AI_WORKFLOW_MODE, resolveAiWorkflowMode, type AiWorkflowMode, type LegacyAiWorkflowMode } from "@/lib/agent/workflow-mode"
 import { DEFAULT_AI_WORKFLOW_MODE, resolveAiWorkflowMode, type AiWorkflowMode, type LegacyAiWorkflowMode } from "@/lib/agent/workflow-mode"
 import { DEFAULT_MCP_CONFIG, type McpConfig } from "@/lib/mcp/config"
 import { DEFAULT_MCP_CONFIG, type McpConfig } from "@/lib/mcp/config"
+import {
+  normalizeProviderConfigs,
+  normalizeUserLlmConfig,
+} from "@/lib/llm-context-size"
 
 
 const GRAPH_LABEL_MODE_KEY = "lk-graph-label-display-mode"
 const GRAPH_LABEL_MODE_KEY = "lk-graph-label-display-mode"
 const GRAPH_EDGE_COLOR_KEY = "lk-graph-edge-color"
 const GRAPH_EDGE_COLOR_KEY = "lk-graph-edge-color"
@@ -140,7 +144,11 @@ interface LlmConfig {
   customEndpoint: string
   customEndpoint: string
   azureApiVersion?: string
   azureApiVersion?: string
   azureModelFamily?: AzureModelFamily
   azureModelFamily?: AzureModelFamily
-  maxContextSize: number // max context window in characters
+  /** The model's context window, in TOKENS, as published on its spec sheet. */
+  maxContextSize: number
+  /** The model's maximum output, in TOKENS. A capability ceiling, not a
+   *  per-request size: what actually gets sent is min(workflow need, this). */
+  maxOutputTokens?: number
   apiMode?: CustomApiMode
   apiMode?: CustomApiMode
   reasoning?: ReasoningConfig
   reasoning?: ReasoningConfig
   localCliIsolation?: boolean
   localCliIsolation?: boolean
@@ -443,6 +451,7 @@ export interface ProviderOverride {
   azureModelFamily?: AzureModelFamily
   azureModelFamily?: AzureModelFamily
   apiMode?: CustomApiMode
   apiMode?: CustomApiMode
   maxContextSize?: number
   maxContextSize?: number
+  maxOutputTokens?: number
   reasoning?: ReasoningConfig
   reasoning?: ReasoningConfig
   localCliIsolation?: boolean
   localCliIsolation?: boolean
   codexCliTimeoutMinutes?: number
   codexCliTimeoutMinutes?: number
@@ -902,14 +911,16 @@ export const useWikiStore = create<WikiState>((set) => ({
   visualStyle: readStoredVisualStyle(),
   visualStyle: readStoredVisualStyle(),
   sidebarNavConfig: readStoredSidebarNavConfig(),
   sidebarNavConfig: readStoredSidebarNavConfig(),
 
 
-  setLlmConfig: (llmConfig) => set({ llmConfig }),
+  setLlmConfig: (llmConfig) => set({ llmConfig: normalizeUserLlmConfig(llmConfig) }),
   setAiChatModel: (aiChatModel) => set({ aiChatModel }),
   setAiChatModel: (aiChatModel) => set({ aiChatModel }),
   setAiOutlineModel: (aiOutlineModel) => set((state) => ({
   setAiOutlineModel: (aiOutlineModel) => set((state) => ({
     aiOutlineModel,
     aiOutlineModel,
     aiOutlineModelRevision: state.aiOutlineModelRevision + 1,
     aiOutlineModelRevision: state.aiOutlineModelRevision + 1,
   })),
   })),
   setDefaultLlmModel: (defaultLlmModel) => set({ defaultLlmModel }),
   setDefaultLlmModel: (defaultLlmModel) => set({ defaultLlmModel }),
-  setProviderConfigs: (providerConfigs) => set({ providerConfigs }),
+  setProviderConfigs: (providerConfigs) => set({
+    providerConfigs: normalizeProviderConfigs(providerConfigs),
+  }),
   setActivePresetId: (activePresetId) => set({ activePresetId }),
   setActivePresetId: (activePresetId) => set({ activePresetId }),
   setSearchApiConfig: (searchApiConfig) => set({ searchApiConfig }),
   setSearchApiConfig: (searchApiConfig) => set({ searchApiConfig }),
   setMcpConfig: (mcpConfig) => set({ mcpConfig }),
   setMcpConfig: (mcpConfig) => set({ mcpConfig }),