浏览代码

fix(llm): 统一输出预算为窗口比例并始终外发 max_tokens

HTTP 供应商按 analysis 0.04 / generation 0.15 规划输出,深章正文保留 15360 地板;
去掉 Anthropic 4096 与 ingest 阶梯特判,并更新 2026-08 preset 模型列表。

Co-authored-by: Cursor <cursoragent@cursor.com>
darknessomi 1 月之前
父节点
当前提交
9303b441ed

+ 59 - 67
src/components/settings/llm-presets.ts

@@ -81,19 +81,17 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     label: "Anthropic (Claude)",
     hint: "Official Claude API",
     provider: "anthropic",
-    defaultModel: "claude-sonnet-4-5-20250929",
-    // Cross-referenced with hermes-agent/hermes_cli/models.py:233-242.
-    // Both shortened and dated aliases work on api.anthropic.com.
+    defaultModel: "claude-sonnet-5",
+    // 2026-08 lineup from platform.claude.com model overview. Retired
+    // 3.5 / Sonnet-4 dated IDs dropped; type them manually if still needed.
     suggestedModels: [
+      "claude-opus-5",
+      "claude-sonnet-5",
+      "claude-opus-4-8",
       "claude-opus-4-7",
       "claude-opus-4-6",
       "claude-sonnet-4-6",
-      "claude-sonnet-4-5-20250929",
-      "claude-haiku-4-5-20251001",
-      "claude-opus-4-5-20251101",
-      "claude-sonnet-4-20250514",
-      "claude-3-5-sonnet-20241022",
-      "claude-3-5-haiku-20241022",
+      "claude-haiku-4-5",
     ],
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
@@ -102,16 +100,18 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     label: "Claude Code CLI (local)",
     hint: "Uses the local `claude` binary — no API key needed",
     provider: "claude-code",
-    defaultModel: "claude-sonnet-4-6",
+    defaultModel: "claude-sonnet-5",
     // Mirrors anthropic preset; the CLI forwards to the same Anthropic
     // backend, so model ids are identical. Users with a subscription
     // can pick Opus/Sonnet/Haiku here without paying an API key bill.
     suggestedModels: [
+      "claude-opus-5",
+      "claude-sonnet-5",
+      "claude-opus-4-8",
       "claude-opus-4-7",
       "claude-opus-4-6",
       "claude-sonnet-4-6",
-      "claude-sonnet-4-5-20250929",
-      "claude-haiku-4-5-20251001",
+      "claude-haiku-4-5",
     ],
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
@@ -156,38 +156,34 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     label: "OpenAI (GPT)",
     hint: "Official OpenAI API",
     provider: "openai",
-    defaultModel: "gpt-4o",
-    // Current public GPT models on api.openai.com. Reasoning models and
-    // the 4.1 family are both exposed under the chat/completions route.
+    defaultModel: "gpt-5.5",
+    // 2026-08 api.openai.com lineup (GPT-5.5 / 5.4 family). Context preset
+    // stops at the 272K long-context pricing inflection; users can raise it.
     suggestedModels: [
-      "gpt-4o",
-      "gpt-4o-mini",
-      "gpt-4.1",
-      "gpt-4.1-mini",
-      "gpt-4.1-nano",
-      "o3",
-      "o3-mini",
-      "o1",
-      "o1-mini",
-      "gpt-4-turbo",
+      "gpt-5.5",
+      "gpt-5.5-pro",
+      "gpt-5.4",
+      "gpt-5.4-pro",
+      "gpt-5.4-mini",
+      "gpt-5.4-nano",
+      "gpt-5.2",
     ],
-    suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
+    suggestedContextSize: 272_000,
   },
   {
     id: "google",
     label: "Google (Gemini)",
     hint: "Generative Language API",
     provider: "google",
-    defaultModel: "gemini-2.5-flash",
-    // 2.5 generation is the current stable; 2.0 kept as fallback.
+    defaultModel: "gemini-3.6-flash",
+    // 2026-08 Gemini API: Flash line is at 3.6; Pro flagship is still the
+    // 3.1 preview — gemini-3.5-pro / 3.6-pro are not published yet.
     suggestedModels: [
-      "gemini-2.5-pro",
-      "gemini-2.5-flash",
-      "gemini-2.5-flash-lite",
-      "gemini-2.0-flash",
-      "gemini-2.0-flash-lite",
-      "gemini-1.5-pro",
-      "gemini-1.5-flash",
+      "gemini-3.6-flash",
+      "gemini-3.5-flash",
+      "gemini-3.5-flash-lite",
+      "gemini-3.1-pro-preview",
+      "gemini-3.1-flash-lite",
     ],
     suggestedContextSize: 1000000,
   },
@@ -198,6 +194,8 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     provider: "azure",
     baseUrl: "https://your-resource.openai.azure.com",
     defaultModel: "your-deployment-name",
+    // Still on the deployment-style URL builder; keep a GA date-based
+    // api-version that those endpoints accept. Foundry v1 is a different path.
     azureApiVersion: "2024-10-21",
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
@@ -258,17 +256,15 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     hint: "api.groq.com",
     provider: "custom",
     baseUrl: "https://api.groq.com/openai/v1",
-    defaultModel: "llama-3.3-70b-versatile",
+    defaultModel: "openai/gpt-oss-120b",
     apiMode: "chat_completions",
-    // Writing workflows require at least 204800 tokens of context.
+    // 2026-08: llama-3.3-70b / llama-3.1-8b shut down 2026-08-16; qwen3-32b
+    // and llama-4-scout already retired. Prefer gpt-oss / qwen3.6 / MiniMax.
     suggestedModels: [
-      "llama-3.3-70b-versatile",
-      "llama-3.1-8b-instant",
-      "llama-3.1-70b-versatile",
-      "moonshotai/kimi-k2-instruct",
       "openai/gpt-oss-120b",
       "openai/gpt-oss-20b",
-      "qwen/qwen3-32b",
+      "qwen/qwen3.6-27b",
+      "minimaxai/minimax-m2.7",
     ],
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
@@ -278,17 +274,17 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     hint: "api.x.ai",
     provider: "custom",
     baseUrl: "https://api.x.ai/v1",
-    defaultModel: "grok-3",
+    defaultModel: "grok-4.5",
     apiMode: "chat_completions",
+    // Live GET /v1/language-models (2026-07) returns exactly these six text
+    // models. Trap: `grok-latest` aliases grok-4.3, not grok-4.5.
     suggestedModels: [
-      "grok-4-latest",
-      "grok-4",
-      "grok-3",
-      "grok-3-mini",
-      "grok-3-fast",
-      "grok-3-mini-fast",
-      "grok-code-fast-1",
-      "grok-2-vision-1212",
+      "grok-4.5",
+      "grok-4.3",
+      "grok-4.20-0309-reasoning",
+      "grok-4.20-0309-non-reasoning",
+      "grok-4.20-multi-agent-0309",
+      "grok-build-0.1",
     ],
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
@@ -305,27 +301,21 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     // per-user from build.nvidia.com. Full catalog is huge and
     // changes often — this is a practical subset; users can type any
     // other id into the custom input.
-    defaultModel: "meta/llama-3.3-70b-instruct",
+    defaultModel: "meta/llama-4-maverick-17b-128e-instruct",
     suggestedModels: [
-      // NVIDIA's own reasoning / agentic models
       "nvidia/llama-3.3-nemotron-super-49b-v1.5",
-      "nvidia/nemotron-3-super-120b-a12b",
       "nvidia/nemotron-3-nano-30b-a3b",
-      // Meta Llama family
+      "meta/llama-4-maverick-17b-128e-instruct",
       "meta/llama-3.3-70b-instruct",
       "meta/llama-3.1-405b-instruct",
-      "meta/llama-3.1-70b-instruct",
-      // Popular third-party agentic / open-weight
-      "deepseek-ai/deepseek-v3.2",
+      "deepseek-ai/deepseek-v4-pro",
+      "deepseek-ai/deepseek-v4-flash",
       "moonshotai/kimi-k2.6",
       "qwen/qwen3.5-397b-a17b",
+      "qwen/qwen3-coder-480b-a35b-instruct",
       "minimaxai/minimax-m2.7",
-      "minimaxai/minimax-m2.5",
-      "z-ai/glm5",
+      "mistralai/mistral-large-3-675b-instruct-2512",
       "openai/gpt-oss-120b",
-      // Mistral family
-      "mistralai/mixtral-8x22b-instruct",
-      "mistralai/mistral-large-2-instruct",
     ],
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },
@@ -548,13 +538,15 @@ const RAW_LLM_PRESETS: LlmPreset[] = [
     provider: "custom",
     baseUrl: "https://ollama.com/v1",
     apiMode: "chat_completions",
-    // Ollama Cloud catalog rotates frequently — keep short common picks.
+    // 2026-08 cloud catalog: deepseek-v3.1 / kimi-k2:1t / qwen3-coder:480b
+    // retired in favor of v4 / k2.6 / qwen3.5.
     suggestedModels: [
       "gpt-oss:120b",
       "gpt-oss:20b",
-      "qwen3-coder:480b",
-      "kimi-k2:1t",
-      "deepseek-v3.1:671b",
+      "deepseek-v4-flash",
+      "deepseek-v4-pro",
+      "qwen3.5:397b",
+      "kimi-k2.6",
     ],
     suggestedContextSize: MIN_USER_LLM_CONTEXT_SIZE,
   },

+ 1 - 1
src/components/settings/llm-wiki-model-settings.spec.ts

@@ -72,7 +72,7 @@ describe("QMAI model settings", () => {
     const groq = preset("groq")
     expect(groq.suggestedModels).not.toContain("mixtral-8x7b-32768")
     expect(groq.suggestedModels).not.toContain("gemma2-9b-it")
-    expect(groq.defaultModel).toBe("llama-3.3-70b-versatile")
+    expect(groq.defaultModel).toBe("openai/gpt-oss-120b")
   })
 
   it("resolves DeepSeek to an OpenAI-compatible custom endpoint", () => {

+ 15 - 16
src/lib/context-budget.test.ts

@@ -142,47 +142,46 @@ describe("outline request budget", () => {
 })
 
 describe("chapter request budget", () => {
-  it.each([
-    [2_000, 8_000],
-    [3_000, 8_000],
-    [6_000, 13_000],
-  ])("derives generation output for %i target chars", (chapterTargetChars, outputTokens) => {
+  it("uses window-fraction generation output with a 15360 floor", () => {
     const plan = planChapterRequestBudget({
       maxContextSize: 204_800,
-      chapterTargetChars,
+      chapterTargetChars: 3_000,
       stage: "generation",
     })
-    expect(plan.outputTokens).toBe(outputTokens)
+    // 0.15 × 204800 = 30720, already above the 15360 floor.
+    expect(plan.outputTokens).toBe(30_720)
     expect(
       plan.outputTokens + plan.contextTokenBudget + plan.scaffoldReserveTokens,
     ).toBeLessThanOrEqual(plan.windowTokens)
   })
 
-  it("keeps chapter output tied to target length, not to the window", () => {
-    // A 3000-character chapter needs the same output on a 1M model as on a 200K
-    // one, so the generous window must not inflate the request.
+  it("scales generation output with the window past the floor", () => {
     expect(planChapterRequestBudget({
       maxContextSize: 1_000_000,
       chapterTargetChars: 3_000,
       stage: "generation",
-    }).outputTokens).toBe(8_000)
+    }).outputTokens).toBe(150_000)
   })
 
-  it("bounds the generation output by the declared output cap", () => {
+  it("lets the declared output cap win over the generation floor", () => {
     expect(planChapterRequestBudget({
       maxContextSize: 204_800,
       chapterTargetChars: 6_000,
       stage: "generation",
-      maxOutputTokens: 4_096,
-    }).outputTokens).toBe(4_096)
+      maxOutputTokens: 8_192,
+    }).outputTokens).toBe(8_192)
   })
 
-  it("uses 4096 tokens for task analysis", () => {
+  it("uses analysis fraction of the window for task analysis", () => {
     expect(planChapterRequestBudget({
       maxContextSize: 204_800,
       chapterTargetChars: 3_000,
       stage: "analysis",
-    }).outputTokens).toBe(4_096)
+    }).outputTokens).toBe(8_192)
+    expect(planChapterRequestBudget({
+      maxContextSize: 1_000_000,
+      stage: "analysis",
+    }).outputTokens).toBe(40_000)
   })
 })
 

+ 22 - 21
src/lib/context-budget.ts

@@ -152,6 +152,16 @@ export function resolveContextPackTokenBudget(
 
 export const MIN_LLM_OUTPUT_TOKENS = 512
 
+/** Share of the window reserved for analysis / planning responses (task brief, etc.). */
+export const ANALYSIS_OUTPUT_FRAC = 0.04
+
+/**
+ * Floor for chapter-generation output budgets, ≈1.5 万字 at the CJK
+ * 1-char-per-token estimator. Window-fraction planning may raise this, but
+ * never drops below it — unless the user's maxOutputTokens cap is lower.
+ */
+export const CHAPTER_GENERATION_OUTPUT_FLOOR = 15_360
+
 export class LlmContextBudgetError extends Error {
   constructor(message = "模型上下文不足:无法同时容纳系统提示、当前用户请求和最小输出空间。") {
     super(message)
@@ -257,13 +267,6 @@ export interface PlanChapterRequestBudgetInput {
   thinkingFloorTokens?: number
 }
 
-function chapterMaxOutputTokens(targetChars?: number): number {
-  const target = Number.isFinite(targetChars) && (targetChars as number) > 0
-    ? Math.max(2_000, Math.min(6_000, Math.round(targetChars as number)))
-    : 3_000
-  return target === 3_000 ? 8_000 : Math.max(8_000, Math.ceil((target + 500) * 2))
-}
-
 export function planChapterRequestBudget(
   input: PlanChapterRequestBudgetInput,
 ): LlmRequestBudgetPlan {
@@ -272,14 +275,18 @@ export function planChapterRequestBudget(
     normalizedWindow,
     input.contextTokenBudget,
   )
-  // Chapter output is sized from the user's target chapter length rather than
-  // a share of the window: a 3000-character chapter needs the same output on a
-  // 200K model as on a 1M one.
+  // Analysis and generation both scale with the window. Generation also keeps
+  // a 15_360-token floor so a short target-char setting cannot starve the
+  // draft; the user's maxOutputTokens cap still wins when it is lower.
+  const desiredOutputTokens = input.stage === "analysis"
+    ? Math.floor(normalizedWindow * ANALYSIS_OUTPUT_FRAC)
+    : Math.max(
+      CHAPTER_GENERATION_OUTPUT_FLOOR,
+      Math.floor(normalizedWindow * RESPONSE_RESERVE_FRAC),
+    )
   return planLlmRequestBudget({
     maxContextSize: normalizedWindow,
-    desiredOutputTokens: input.stage === "analysis"
-      ? 4_096
-      : chapterMaxOutputTokens(input.chapterTargetChars),
+    desiredOutputTokens,
     requestedContextTokens: genericContextCap,
     scaffoldReserveTokens: 8_000,
     minimumContextTokens: 2_000,
@@ -290,12 +297,6 @@ export function planChapterRequestBudget(
 
 export type OutlineBudgetStage = "analysis" | "generation"
 
-/** Share of the window the outline's own response may claim. Reuses the
- *  response reserve the rest of the budgeting layer already assumes. */
-const OUTLINE_GENERATION_OUTPUT_FRAC = RESPONSE_RESERVE_FRAC
-/** Analysis passes summarise rather than draft, so they need far less. */
-const OUTLINE_ANALYSIS_OUTPUT_FRAC = 0.04
-
 export interface PlanOutlineRequestBudgetInput {
   maxContextSize?: number
   contextTokenBudget?: number
@@ -311,8 +312,8 @@ export function planOutlineRequestBudget(
   // Scales with the window instead of stepping through fixed tiers, and is
   // then bounded by the user's declared output cap inside the kernel.
   const desiredOutputTokens = Math.floor(normalizedWindow * (input.stage === "analysis"
-    ? OUTLINE_ANALYSIS_OUTPUT_FRAC
-    : OUTLINE_GENERATION_OUTPUT_FRAC))
+    ? ANALYSIS_OUTPUT_FRAC
+    : RESPONSE_RESERVE_FRAC))
   const genericContextCap = computeNovelContextTokenBudget(
     normalizedWindow,
     input.contextTokenBudget,

+ 10 - 15
src/lib/ingest.prompt.test.ts

@@ -12,25 +12,20 @@ import {
 // 1 = CJK) so they stay deterministic regardless of the active UI language.
 describe("long-source ingest planning", () => {
   it("scales generation output tokens with the configured context window", () => {
-    expect(computeIngestGenerationMaxTokens(64_000)).toBe(8_192)
-    expect(computeIngestGenerationMaxTokens(128_000)).toBe(16_384)
-    expect(computeIngestGenerationMaxTokens(256_000)).toBe(24_576)
-    expect(computeIngestGenerationMaxTokens(1_000_000)).toBe(32_768)
-    expect(computeIngestReviewMaxTokens(1_000_000)).toBe(8_192)
+    // Windows below MIN_USER_LLM_CONTEXT_SIZE (204800) clamp up first.
+    expect(computeIngestGenerationMaxTokens(64_000)).toBe(30_720)
+    expect(computeIngestGenerationMaxTokens(204_800)).toBe(30_720)
+    expect(computeIngestGenerationMaxTokens(1_000_000)).toBe(150_000)
+    expect(computeIngestReviewMaxTokens(1_000_000)).toBe(40_000)
   })
 
-  it("picks the output tier from the token window alone", () => {
-    // A model's output ceiling is a property of the model, not of the UI
-    // language, so the tier no longer moves with character density.
-    expect(computeIngestGenerationMaxTokens(128_000)).toBe(16_384)
+  it("uses the same window-fraction formula regardless of UI language density", () => {
+    expect(computeIngestGenerationMaxTokens(256_000)).toBe(38_400)
   })
 
-  it("scales analysis output tokens with the window but caps at 8192 (floor 4096)", () => {
-    // Small window keeps the legacy 4096 floor.
-    expect(computeIngestAnalysisMaxTokens(64_000)).toBe(4_096)
-    // Larger windows scale up but never exceed the 8192 cap.
-    expect(computeIngestAnalysisMaxTokens(128_000)).toBe(8_192)
-    expect(computeIngestAnalysisMaxTokens(1_000_000)).toBe(8_192)
+  it("scales analysis output tokens with ANALYSIS_OUTPUT_FRAC", () => {
+    expect(computeIngestAnalysisMaxTokens(204_800)).toBe(8_192)
+    expect(computeIngestAnalysisMaxTokens(1_000_000)).toBe(40_000)
   })
 
   it("scales source budget from the configured context window instead of a fixed 50k cap", () => {

+ 19 - 27
src/lib/ingest.ts

@@ -6,7 +6,12 @@ import {
   writeFile,
   listDirectory,
 } from "@/commands/fs"
-import { computeContextBudget } from "@/lib/context-budget"
+import {
+  ANALYSIS_OUTPUT_FRAC,
+  RESPONSE_RESERVE_FRAC,
+  computeContextBudget,
+} from "@/lib/context-budget"
+import { normalizeUserLlmContextSize } from "@/lib/llm-context-size"
 import { streamChat } from "@/lib/llm-client"
 import type { LlmConfig } from "@/stores/wiki-store"
 import { useWikiStore } from "@/stores/wiki-store"
@@ -65,10 +70,6 @@ const LONG_SOURCE_CHUNK_MIN = 12_000
 const LONG_SOURCE_CHUNK_MAX = 60_000
 const LONG_SOURCE_DIGEST_MAX = 15_000
 const LONG_SOURCE_CHUNK_ANALYSIS_MAX = 40_000
-const INGEST_GENERATION_TOKENS_DEFAULT = 8_192
-const INGEST_GENERATION_TOKENS_128K = 16_384
-const INGEST_GENERATION_TOKENS_256K = 24_576
-const INGEST_GENERATION_TOKENS_512K = 32_768
 const REVIEW_STAGE_MIN_SIGNAL_CHARS = 10_000
 const REVIEW_STAGE_MIN_FILE_BLOCKS = 4
 
@@ -1231,46 +1232,37 @@ export function computeIngestSourceBudget(
 }
 
 /**
- * Output ladder for wiki page generation, stepped off the model's token
- * window. Compares the window directly rather than a language-scaled
- * character budget: a model's output ceiling does not shrink because the UI
- * is in Chinese, and the old comparison dropped CJK users a whole tier.
+ * Output budget for wiki page generation: window × RESPONSE_RESERVE_FRAC.
+ * Same formula every other long-form path uses; fitIngestOutputToWindow still
+ * clamps against remaining room after the packed prompt.
  */
 export function computeIngestGenerationMaxTokens(
   maxContextSize: number | undefined,
 ): number {
-  const windowTokens = typeof maxContextSize === "number" && maxContextSize > 0
-    ? maxContextSize
-    : DEFAULT_INGEST_WINDOW_TOKENS
-  if (windowTokens >= 512_000) return INGEST_GENERATION_TOKENS_512K
-  if (windowTokens >= 256_000) return INGEST_GENERATION_TOKENS_256K
-  if (windowTokens >= 128_000) return INGEST_GENERATION_TOKENS_128K
-  return INGEST_GENERATION_TOKENS_DEFAULT
+  const windowTokens = normalizeUserLlmContextSize(maxContextSize)
+  return Math.max(INGEST_OUTPUT_TOKEN_FLOOR, Math.floor(windowTokens * RESPONSE_RESERVE_FRAC))
 }
 
 export function computeIngestReviewMaxTokens(
   maxContextSize: number | undefined,
 ): number {
-  return Math.min(8_192, Math.max(4_096, Math.floor(computeIngestGenerationMaxTokens(maxContextSize) / 2)))
+  const windowTokens = normalizeUserLlmContextSize(maxContextSize)
+  return Math.max(INGEST_OUTPUT_TOKEN_FLOOR, Math.floor(windowTokens * ANALYSIS_OUTPUT_FRAC))
 }
 
 /**
- * Output-token budget for the intermediate analysis passes (whole-source
- * analysis and per-chunk long-source analysis). Previously hard-coded to
- * 4096; now scales off the generation ladder so larger context windows get
- * a richer analysis, while staying capped well below a full page-generation
- * pass. Small windows retain the original 4096 floor.
+ * Output-token budget for intermediate analysis passes (whole-source and
+ * per-chunk). Uses the shared analysis fraction of the window.
  */
 export function computeIngestAnalysisMaxTokens(
   maxContextSize: number | undefined,
 ): number {
-  return Math.min(8_192, Math.max(4_096, Math.floor(computeIngestGenerationMaxTokens(maxContextSize) / 2)))
+  const windowTokens = normalizeUserLlmContextSize(maxContextSize)
+  return Math.max(INGEST_OUTPUT_TOKEN_FLOOR, Math.floor(windowTokens * ANALYSIS_OUTPUT_FRAC))
 }
 
 /** chars/token the ingest budgeting assumes; mirrors context-budget.ts. */
 const INGEST_CHARS_PER_TOKEN = 4
-/** Window assumed when the config carries none; mirrors context-budget.ts. */
-const DEFAULT_INGEST_WINDOW_TOKENS = 204_800
 /** Smallest output allowance we will still request when the window is nearly
  *  full — below this a response is useless, so we accept a tiny overflow risk
  *  rather than emitting nothing. */
@@ -1278,8 +1270,8 @@ const INGEST_OUTPUT_TOKEN_FLOOR = 512
 
 /**
  * Clamp a desired output-token count so that (packed prompt + output) fits the
- * model's real token window. `desiredTokens` is the ladder value; we only ever
- * reduce it when the prompt already leaves less room than the ladder wants.
+ * model's real token window. `desiredTokens` is the window-fraction budget; we
+ * only ever reduce it when the prompt already leaves less room than that.
  *
  * Language-aware: CJK text is denser, so the same prompt consumes more real
  * tokens and leaves less room for output. The English-calibrated window (4:1)

+ 77 - 21
src/lib/llm-client.ts

@@ -158,6 +158,32 @@ function inputLengthLimitMessage(limit: { inputLength: number; maxLength: number
   return `输入内容过长:本次请求约 ${limit.inputLength} 字符,接口最大允许 ${limit.maxLength} 字符。请减少历史上下文、缩短章节正文,或确认当前接口是否真的支持所选模型的上下文长度。`
 }
 
+/**
+ * Detect provider rejections of an oversized max_tokens / max_output_tokens
+ * request. Returns the highest value the model will accept when the error
+ * reports one; otherwise null.
+ */
+export function parseMaxTokensLimit(errorDetail: string): number | null {
+  const patterns = [
+    /max[_ ]?(?:output[_ ]?)?tokens?\s*(?:of\s+|is\s+|=\s*)?([\d,]+)\s*(?:is\s+)?(?:too\s+(?:large|high)|exceeds|above|greater than)/i,
+    /(?:max[_ ]?(?:output[_ ]?)?tokens?|maximum\s+output)\s*(?:must be|should be|limited to|capped at|<=|≤)\s*([\d,]+)/i,
+    /(?:supports?|allows?|accepts?)\s+(?:at most|up to|a maximum of)\s*([\d,]+)\s*(?:output\s+)?tokens?/i,
+    /max[_ ]?(?:completion|output)[_ ]?tokens?\s*(?:cannot exceed|must not exceed|<=|≤)\s*([\d,]+)/i,
+    /(?:this model|model)\s+(?:has a )?(?:maximum|max)\s+(?:of\s+)?([\d,]+)\s*(?:output\s+)?tokens?/i,
+  ]
+  for (const pattern of patterns) {
+    const match = pattern.exec(errorDetail)
+    if (!match) continue
+    const value = Number(match[1]?.replace(/,/g, ""))
+    if (Number.isFinite(value) && value >= 512) return Math.floor(value)
+  }
+  return null
+}
+
+function isLocalCliProvider(provider: LlmConfig["provider"]): boolean {
+  return provider === "claude-code" || provider === "codex-cli" || provider === "cursor-cli"
+}
+
 export async function streamChat(
   config: LlmConfig,
   messages: import("./llm-providers").ChatMessage[],
@@ -181,9 +207,6 @@ export async function streamChat(
   const thinkingFloorTokens = thinkingMinMaxTokens(runtimeConfig.reasoning ?? { mode: "auto" })
   const runtimeBudget = planLlmRequestBudget({
     maxContextSize: configuredWindow,
-    // Without an explicit request the response reserve is only used to size
-    // the INPUT trim; it is not sent as max_tokens unless thinking needs it
-    // (see shouldSendMaxTokens below).
     desiredOutputTokens: requestOverrides?.max_tokens
       ?? Math.floor(configuredWindow * RESPONSE_RESERVE_FRAC),
     scaffoldReserveTokens: toolScaffoldTokens,
@@ -192,12 +215,10 @@ export async function streamChat(
     thinkingFloorTokens,
   })
   let effectiveOutputTokens = runtimeBudget.outputTokens
-  // Emit max_tokens when the caller asked for one, or when explicit thinking
-  // needs a known output allowance (otherwise OpenAI-compatible paths keep
-  // thinking on against an unknown provider default). auto/off without a
-  // caller value still omits the field so long-form keeps the provider default.
-  let shouldSendMaxTokens =
-    requestOverrides?.max_tokens !== undefined || thinkingFloorTokens > 0
+  // HTTP providers always receive the planned max_tokens so every vendor
+  // follows the same window-fraction rule. Local CLI transports have no
+  // equivalent flag and warn in DEV when the field is present.
+  const shouldSendMaxTokens = !isLocalCliProvider(runtimeConfig.provider)
   let budgetedMessages: import("./llm-providers").ChatMessage[]
   try {
     budgetedMessages = trimChatMessagesToTokenBudget(
@@ -224,13 +245,13 @@ export async function streamChat(
           - estimateChatMessagesTokens(budgetedMessages),
       ),
     )
-    // The input was trimmed against a reserved output slot, so that slot has
-    // to be declared even if the caller never asked for one.
-    shouldSendMaxTokens = true
   }
-  const effectiveRequestOverrides: RequestOverrides = shouldSendMaxTokens
+  let effectiveRequestOverrides: RequestOverrides = shouldSendMaxTokens
     ? { ...requestOverrides, max_tokens: effectiveOutputTokens }
-    : { ...requestOverrides }
+    : (() => {
+      const { max_tokens: _ignored, ...rest } = requestOverrides ?? {}
+      return rest
+    })()
   const { onToken, onDone, onError } = callbacks
   const decoder = new TextDecoder()
 
@@ -282,10 +303,13 @@ export async function streamChat(
   }
 
   try {
-    const buildRequestInit = (nextMessages: import("./llm-providers").ChatMessage[]): RequestInit => ({
+    const buildRequestInit = (
+      nextMessages: import("./llm-providers").ChatMessage[],
+      overrides: RequestOverrides = effectiveRequestOverrides,
+    ): RequestInit => ({
       method: "POST",
       headers: providerConfig.headers,
-      body: JSON.stringify(providerConfig.buildBody(nextMessages, effectiveRequestOverrides)),
+      body: JSON.stringify(providerConfig.buildBody(nextMessages, overrides)),
       signal: combinedSignal,
     })
 
@@ -379,7 +403,7 @@ export async function streamChat(
         // Surface probe in the toast/UI error — console.warn alone stays in WebView DevTools.
         errorDetail += `\n${formatReasoningReplayRiskForError(risk)}`
       }
-      let inputLimitRetrySucceeded = false
+      let httpRetrySucceeded = false
       const inputLimit = parseInputLengthLimit(errorDetail)
       if (inputLimit) {
         const currentInputTokens = estimateChatMessagesTokens(budgetedMessages)
@@ -414,7 +438,7 @@ export async function streamChat(
           return
         }
         if (response.ok) {
-          inputLimitRetrySucceeded = true
+          httpRetrySucceeded = true
         } else {
           let retryErrorDetail = `HTTP ${response.status}: ${response.statusText}`
           try {
@@ -427,8 +451,40 @@ export async function streamChat(
           return
         }
       }
+      const reportedMaxTokens = !httpRetrySucceeded ? parseMaxTokensLimit(errorDetail) : null
+      if (
+        reportedMaxTokens !== null
+        && typeof effectiveRequestOverrides.max_tokens === "number"
+        && reportedMaxTokens < effectiveRequestOverrides.max_tokens
+      ) {
+        effectiveOutputTokens = reportedMaxTokens
+        effectiveRequestOverrides = {
+          ...effectiveRequestOverrides,
+          max_tokens: reportedMaxTokens,
+        }
+        requestInit = buildRequestInit(budgetedMessages, effectiveRequestOverrides)
+        try {
+          response = await sendRequest(requestInit)
+        } catch (err) {
+          onError(err instanceof Error ? err : new Error(String(err)))
+          return
+        }
+        if (response.ok) {
+          httpRetrySucceeded = true
+        } else {
+          let retryErrorDetail = `HTTP ${response.status}: ${response.statusText}`
+          try {
+            const retryBody = await response.text()
+            if (retryBody) retryErrorDetail += ` — ${retryBody}`
+          } catch {
+            // ignore body read failure
+          }
+          onError(new Error(retryErrorDetail))
+          return
+        }
+      }
       if (
-        !inputLimitRetrySucceeded &&
+        !httpRetrySucceeded &&
         response.status === 404 &&
         (runtimeConfig.provider === "azure" ||
           (runtimeConfig.provider === "custom" && isAzureOpenAiEndpoint(runtimeConfig.customEndpoint)))
@@ -440,7 +496,7 @@ export async function streamChat(
         )
         return
       }
-      if (!inputLimitRetrySucceeded && shouldRetryWithBrowserFetch(errorDetail) && typeof globalThis.fetch === "function") {
+      if (!httpRetrySucceeded && shouldRetryWithBrowserFetch(errorDetail) && typeof globalThis.fetch === "function") {
         try {
           response = await globalThis.fetch(providerConfig.url, requestInit)
         } catch (err) {
@@ -459,7 +515,7 @@ export async function streamChat(
           onError(new Error(retryErrorDetail))
           return
         }
-      } else if (!inputLimitRetrySucceeded) {
+      } else if (!httpRetrySucceeded) {
         onError(new Error(errorDetail))
         return
       }

+ 128 - 19
src/lib/llm-client.usage.spec.ts

@@ -5,7 +5,6 @@ import { estimateChatMessagesTokens } from "./chat-request-budget"
 import type { ChatMessage } from "./llm-providers"
 import { thinkingMinMaxTokens } from "./llm-providers"
 import {
-  LlmContextBudgetError,
   RESPONSE_RESERVE_FRAC,
   planLlmRequestBudget,
 } from "./context-budget"
@@ -13,6 +12,7 @@ import { normalizeUserLlmMaxOutputTokens } from "./llm-context-size"
 
 const mocks = vi.hoisted(() => ({
   fetch: vi.fn(),
+  streamClaudeCodeCli: vi.fn(),
 }))
 
 vi.mock("./tauri-fetch", () => ({
@@ -24,6 +24,10 @@ vi.mock("./local-cli-config", () => ({
   resolveRuntimeLocalCliConfig: vi.fn(async (config: LlmConfig) => config),
 }))
 
+vi.mock("./claude-cli-transport", () => ({
+  streamClaudeCodeCli: (...args: unknown[]) => mocks.streamClaudeCodeCli(...args),
+}))
+
 const config: LlmConfig = {
   provider: "openai",
   apiKey: "sk-test",
@@ -36,6 +40,7 @@ const config: LlmConfig = {
 describe("streamChat usage", () => {
   beforeEach(() => {
     mocks.fetch.mockReset()
+    mocks.streamClaudeCodeCli.mockReset()
   })
 
   it("requests and emits OpenAI stream usage once", async () => {
@@ -117,11 +122,15 @@ describe("streamChat usage", () => {
       "",
     ].join("\n"), { status: 200 }))
 
-    // 1843-token window (2048 × 0.9) against ~1800 tokens of CJK input, so the
-    // trim has to bite while leaving the protected messages intact.
-    await streamChat({ ...config, maxContextSize: 2_048 }, [
-      { role: "system", content: "系统".repeat(450) },
-      { role: "user", content: `任务目标:续写。${"正文".repeat(450)}结尾限制:保持人物关系。` },
+    // Window clamps to ≥204800; overflow with a large middle user turn so trim
+    // must drop history while keeping the system + current-user ends intact.
+    const windowTokens = 204_800
+    const outputReserve = Math.floor(windowTokens * RESPONSE_RESERVE_FRAC)
+    await streamChat({ ...config, maxContextSize: windowTokens }, [
+      { role: "system", content: "系统".repeat(20_000) },
+      { role: "user", content: "旧请求".repeat(90_000) },
+      { role: "assistant", content: "旧回复".repeat(90_000) },
+      { role: "user", content: `任务目标:续写。${"正文".repeat(40_000)}结尾限制:保持人物关系。` },
     ], {
       onToken: vi.fn(),
       onDone: vi.fn(),
@@ -133,32 +142,59 @@ describe("streamChat usage", () => {
       messages: ChatMessage[]
       max_tokens?: number
     }
-    expect(estimateChatMessagesTokens(body.messages)).toBeLessThanOrEqual(1_331)
+    expect(estimateChatMessagesTokens(body.messages)).toBeLessThanOrEqual(
+      windowTokens - outputReserve,
+    )
     expect(String(body.messages[0]?.content).trim()).not.toBe("")
     expect(body.messages.at(-1)?.content).toContain("任务目标")
     expect(body.messages.at(-1)?.content).toContain("保持人物关系")
   })
 
-  it("上下文无法容纳最小输出时明确失败且不调用供应商", async () => {
-    await expect(streamChat({ ...config, maxContextSize: 512 }, [
-      { role: "system", content: "系统约束" },
-      { role: "user", content: "生成第一卷完整大纲" },
+  it("超大受保护消息在归一化窗口下会被压缩后仍发送", async () => {
+    // With maxContextSize clamped to ≥204800, the old "tiny window → hard fail"
+    // path is unreachable through streamChat; protected ends are compressed
+    // instead so the request can still leave.
+    mocks.fetch.mockResolvedValue(new Response([
+      'data: {"choices":[{"delta":{"content":"完成"}}]}',
+      "data: [DONE]",
+      "",
+    ].join("\n"), { status: 200 }))
+
+    await streamChat({ ...config, maxContextSize: 204_800 }, [
+      { role: "system", content: "系统".repeat(120_000) },
+      { role: "user", content: "生成".repeat(120_000) },
     ], {
       onToken: vi.fn(),
       onDone: vi.fn(),
       onError: vi.fn(),
-    })).rejects.toBeInstanceOf(LlmContextBudgetError)
+    })
 
-    expect(mocks.fetch).not.toHaveBeenCalled()
+    expect(mocks.fetch).toHaveBeenCalledTimes(1)
+    const body = JSON.parse(String((mocks.fetch.mock.calls[0][1] as RequestInit).body)) as {
+      messages: ChatMessage[]
+    }
+    expect(estimateChatMessagesTokens(body.messages)).toBeLessThan(204_800)
+    expect(String(body.messages[0]?.content).trim()).not.toBe("")
+    expect(String(body.messages.at(-1)?.content).trim()).not.toBe("")
   })
 
-  it("调用方未传 max_tokens 时请求体不带该字段", async () => {
+  it("调用方未传 max_tokens 时仍外发窗口比例预算", async () => {
     mocks.fetch.mockResolvedValue(new Response([
       'data: {"choices":[{"delta":{"content":"完成"}}]}',
       "data: [DONE]",
       "",
     ].join("\n"), { status: 200 }))
 
+    const planned = planLlmRequestBudget({
+      maxContextSize: config.maxContextSize,
+      desiredOutputTokens: Math.floor(
+        Math.max(204_800, config.maxContextSize) * RESPONSE_RESERVE_FRAC,
+      ),
+      scaffoldReserveTokens: 0,
+      minimumContextTokens: 64,
+      maxOutputTokensCap: normalizeUserLlmMaxOutputTokens(config.maxOutputTokens),
+    })
+
     await streamChat(config, [{ role: "user", content: "写第一章" }], {
       onToken: vi.fn(),
       onDone: vi.fn(),
@@ -166,16 +202,28 @@ describe("streamChat usage", () => {
     })
 
     const request = mocks.fetch.mock.calls[0][1] as RequestInit
-    expect(JSON.parse(String(request.body))).not.toHaveProperty("max_tokens")
+    expect(JSON.parse(String(request.body))).toMatchObject({
+      max_tokens: planned.outputTokens,
+    })
   })
 
-  it("reasoning.mode=auto 且调用方未传 max_tokens 时请求体仍省略该字段", async () => {
+  it("reasoning.mode=auto 且调用方未传 max_tokens 时仍外发窗口比例预算", async () => {
     mocks.fetch.mockResolvedValue(new Response([
       'data: {"choices":[{"delta":{"content":"完成"}}]}',
       "data: [DONE]",
       "",
     ].join("\n"), { status: 200 }))
 
+    const planned = planLlmRequestBudget({
+      maxContextSize: config.maxContextSize,
+      desiredOutputTokens: Math.floor(
+        Math.max(204_800, config.maxContextSize) * RESPONSE_RESERVE_FRAC,
+      ),
+      scaffoldReserveTokens: 0,
+      minimumContextTokens: 64,
+      maxOutputTokensCap: normalizeUserLlmMaxOutputTokens(config.maxOutputTokens),
+    })
+
     await streamChat(
       { ...config, reasoning: { mode: "auto" } },
       [{ role: "user", content: "写第一章" }],
@@ -183,7 +231,9 @@ describe("streamChat usage", () => {
     )
 
     const request = mocks.fetch.mock.calls[0][1] as RequestInit
-    expect(JSON.parse(String(request.body))).not.toHaveProperty("max_tokens")
+    expect(JSON.parse(String(request.body))).toMatchObject({
+      max_tokens: planned.outputTokens,
+    })
   })
 
   it("reasoning.mode=high 且调用方未传 max_tokens 时发送预算规划的 max_tokens", async () => {
@@ -196,9 +246,10 @@ describe("streamChat usage", () => {
     const reasoning = { mode: "high" as const }
     const thinkingFloorTokens = thinkingMinMaxTokens(reasoning)
     expect(thinkingFloorTokens).toBeGreaterThan(0)
+    const windowTokens = Math.max(204_800, config.maxContextSize)
     const planned = planLlmRequestBudget({
-      maxContextSize: config.maxContextSize,
-      desiredOutputTokens: Math.floor(config.maxContextSize * RESPONSE_RESERVE_FRAC),
+      maxContextSize: windowTokens,
+      desiredOutputTokens: Math.floor(windowTokens * RESPONSE_RESERVE_FRAC),
       scaffoldReserveTokens: 0,
       minimumContextTokens: 64,
       maxOutputTokensCap: normalizeUserLlmMaxOutputTokens(config.maxOutputTokens),
@@ -237,6 +288,64 @@ describe("streamChat usage", () => {
     expect(JSON.parse(String(request.body))).toMatchObject({ max_tokens: 65_536 })
   })
 
+  it("本地 CLI 供应商不外发 max_tokens", async () => {
+    mocks.streamClaudeCodeCli.mockImplementation(async (
+      _config: LlmConfig,
+      _messages: ChatMessage[],
+      callbacks: { onToken: (token: string) => void; onDone: () => void },
+      _signal?: AbortSignal,
+      overrides?: { max_tokens?: number },
+    ) => {
+      expect(overrides).not.toHaveProperty("max_tokens")
+      callbacks.onToken("ok")
+      callbacks.onDone()
+    })
+
+    await streamChat(
+      {
+        ...config,
+        provider: "claude-code",
+        model: "claude-sonnet-5",
+      },
+      [{ role: "user", content: "写第一章" }],
+      { onToken: vi.fn(), onDone: vi.fn(), onError: vi.fn() },
+      undefined,
+      { max_tokens: 30_720 },
+    )
+
+    expect(mocks.fetch).not.toHaveBeenCalled()
+    expect(mocks.streamClaudeCodeCli).toHaveBeenCalledTimes(1)
+  })
+
+  it("服务商回报 max_tokens 超限时按其上限重试一次", async () => {
+    mocks.fetch
+      .mockResolvedValueOnce(new Response(
+        JSON.stringify({
+          error: {
+            message: "max_tokens is too large: this model supports at most 8192 output tokens",
+          },
+        }),
+        { status: 400 },
+      ))
+      .mockResolvedValueOnce(new Response([
+        'data: {"choices":[{"delta":{"content":"完成"}}]}',
+        "data: [DONE]",
+        "",
+      ].join("\n"), { status: 200 }))
+
+    const onError = vi.fn()
+    await streamChat(config, [{ role: "user", content: "写第一章" }], {
+      onToken: vi.fn(),
+      onDone: vi.fn(),
+      onError,
+    })
+
+    expect(mocks.fetch).toHaveBeenCalledTimes(2)
+    const retryBody = JSON.parse(String((mocks.fetch.mock.calls[1][1] as RequestInit).body))
+    expect(retryBody.max_tokens).toBe(8_192)
+    expect(onError).not.toHaveBeenCalled()
+  })
+
   it("脏 SSE 行不会中断整轮流式响应", async () => {
     const encoder = new TextEncoder()
     const body = new ReadableStream<Uint8Array>({

+ 11 - 0
src/lib/llm-providers.spec.ts

@@ -39,6 +39,17 @@ describe("llm provider reasoning options", () => {
     expect(body.thinking).toEqual({ type: "enabled", budget_tokens: 3_584 })
   })
 
+  it("defaults Anthropic max_tokens to the window-fraction reserve, not 4096", () => {
+    const body = getProviderConfig(customConfig({
+      apiMode: "anthropic_messages",
+      maxContextSize: 204_800,
+    })).buildBody(
+      [{ role: "user", content: "请回答。" }],
+    ) as { max_tokens: number }
+
+    expect(body.max_tokens).toBe(30_720)
+  })
+
   it("does not inflate an Anthropic output budget too small for explicit thinking", () => {
     const body = getProviderConfig(customConfig({
       apiMode: "anthropic_messages",

+ 19 - 4
src/lib/llm-providers.ts

@@ -6,9 +6,10 @@ import {
 } from "@/lib/azure-openai"
 import { normalizeEndpoint } from "@/lib/endpoint-normalizer"
 import {
-  MIN_USER_LLM_CONTEXT_SIZE,
+  normalizeUserLlmContextSize,
   normalizeUserLlmMaxOutputTokens,
 } from "@/lib/llm-context-size"
+import { RESPONSE_RESERVE_FRAC } from "./context-budget"
 import type { LlmUsage } from "./llm-usage"
 import type { UserMemorySurface } from "./user-memory/types"
 
@@ -674,7 +675,7 @@ function isDeepSeekEndpoint(config: LlmConfig): boolean {
  * them.
  */
 export function getEffectiveMaxContextSize(config: LlmConfig): number {
-  return config.maxContextSize || MIN_USER_LLM_CONTEXT_SIZE
+  return normalizeUserLlmContextSize(config.maxContextSize)
 }
 
 /** The declared output ceiling to plan against, in tokens. */
@@ -924,7 +925,21 @@ function buildAnthropicSystem(messages: ChatMessage[]): string | unknown[] | und
   return blocks.length > 0 ? blocks : undefined
 }
 
+function defaultAnthropicMaxTokens(config: LlmConfig): number {
+  // Anthropic requires max_tokens. Prefer the same window-fraction plan the
+  // HTTP kernel uses; never fall back to a vendor-hardcoded 4096.
+  const windowTokens = getEffectiveMaxContextSize(config)
+  return Math.max(
+    512,
+    Math.min(
+      getEffectiveMaxOutputTokens(config),
+      Math.floor(windowTokens * RESPONSE_RESERVE_FRAC),
+    ),
+  )
+}
+
 function buildAnthropicBody(
+  config: LlmConfig,
   messages: ChatMessage[],
   overrides?: RequestOverrides,
 ): Record<string, unknown> {
@@ -941,7 +956,7 @@ function buildAnthropicBody(
     messages: conversationMessages,
     ...(system !== undefined ? { system } : {}),
     stream: true,
-    max_tokens: overrides?.max_tokens ?? 4096,
+    max_tokens: overrides?.max_tokens ?? defaultAnthropicMaxTokens(config),
     ...(overrides?.temperature !== undefined ? { temperature: overrides.temperature } : {}),
     ...(overrides?.top_p !== undefined ? { top_p: overrides.top_p } : {}),
     ...(overrides?.top_k !== undefined ? { top_k: overrides.top_k } : {}),
@@ -956,7 +971,7 @@ function buildAnthropicBodyWithReasoning(
   messages: ChatMessage[],
   overrides?: RequestOverrides,
 ): Record<string, unknown> {
-  const body = buildAnthropicBody(messages, overrides)
+  const body = buildAnthropicBody(config, messages, overrides)
   const reasoning = effectiveReasoning(config, overrides)
   if (reasoning.mode === "auto" || reasoning.mode === "off") return body
 

+ 8 - 3
src/lib/novel/deep-chapter-generation.spec.ts

@@ -1248,8 +1248,9 @@ describe("runDeepChapterGeneration", () => {
 
     expect(overrides.length).toBeGreaterThan(0)
     expect(overrides.every((item) => typeof item?.max_tokens === "number")).toBe(true)
-    expect(overrides.some((item) => item?.max_tokens === 4_096)).toBe(true)
-    expect(overrides.some((item) => item?.max_tokens === 8_000)).toBe(true)
+    // Shared window clamps to 204800: analysis 0.04 → 8192, generation 0.15 → 30720.
+    expect(overrides.some((item) => item?.max_tokens === 8_192)).toBe(true)
+    expect(overrides.some((item) => item?.max_tokens === 30_720)).toBe(true)
   })
 
   it("raises stage output to the floor the configured reasoning level needs", async () => {
@@ -1277,7 +1278,11 @@ describe("runDeepChapterGeneration", () => {
     )
 
     expect(overrides.length).toBeGreaterThan(0)
-    expect(overrides.every((item) => item?.max_tokens === 16_384)).toBe(true)
+    // Analysis is raised to the 16384 thinking floor; generation stays at
+    // the larger window-fraction budget (30720 on the shared 204800 window).
+    expect(overrides.every((item) => (item?.max_tokens ?? 0) >= 16_384)).toBe(true)
+    expect(overrides.some((item) => item?.max_tokens === 16_384)).toBe(true)
+    expect(overrides.some((item) => item?.max_tokens === 30_720)).toBe(true)
   })
 
   it("preserves configured model reasoning for chapter generation calls", async () => {

+ 24 - 10
src/lib/novel/deep-chapter-generation.ts

@@ -669,26 +669,37 @@ export async function runDeepChapterGeneration(
     ? Math.min(...sharedContextWindows)
     : input.llmConfig.maxContextSize;
 
-  // 大纲与其余上下文共用同一窗口预算:按单章目标字数×2预留输出,再分配资料包。
-  const sharedMaxOutputTokens = getEffectiveMaxOutputTokens(input.llmConfig);
-  const sharedThinkingFloor = thinkingMinMaxTokens(
-    input.llmConfig.reasoning ?? { mode: "auto" },
-  );
+  // 大纲与其余上下文共用同一窗口预算:资料包按三阶段最小窗分配。
+  // 输出上限与思考地板按各阶段实际调用的模型重算,不再只读入口 llmConfig。
   const chapterAnalysisBudget = planChapterRequestBudget({
     maxContextSize: sharedContextWindow,
     contextTokenBudget: novelConfig.contextTokenBudget,
     chapterTargetChars: novelConfig.chapterTargetChars,
     stage: "analysis",
-    maxOutputTokens: sharedMaxOutputTokens,
-    thinkingFloorTokens: sharedThinkingFloor,
+    maxOutputTokens: getEffectiveMaxOutputTokens(workflowConfig),
+    thinkingFloorTokens: thinkingMinMaxTokens(
+      workflowConfig.reasoning ?? { mode: "auto" },
+    ),
   });
   const chapterGenerationBudget = planChapterRequestBudget({
     maxContextSize: sharedContextWindow,
     contextTokenBudget: novelConfig.contextTokenBudget,
     chapterTargetChars: novelConfig.chapterTargetChars,
     stage: "generation",
-    maxOutputTokens: sharedMaxOutputTokens,
-    thinkingFloorTokens: sharedThinkingFloor,
+    maxOutputTokens: getEffectiveMaxOutputTokens(writingConfig),
+    thinkingFloorTokens: thinkingMinMaxTokens(
+      writingConfig.reasoning ?? { mode: "auto" },
+    ),
+  });
+  const chapterDeAiBudget = planChapterRequestBudget({
+    maxContextSize: sharedContextWindow,
+    contextTokenBudget: novelConfig.contextTokenBudget,
+    chapterTargetChars: novelConfig.chapterTargetChars,
+    stage: "generation",
+    maxOutputTokens: getEffectiveMaxOutputTokens(deAiConfig),
+    thinkingFloorTokens: thinkingMinMaxTokens(
+      deAiConfig.reasoning ?? { mode: "auto" },
+    ),
   });
   const totalContextTokenBudget = chapterGenerationBudget.contextTokenBudget;
   const analysisRequestOverrides: RequestOverrides = {
@@ -697,6 +708,9 @@ export async function runDeepChapterGeneration(
   const generationRequestOverrides: RequestOverrides = {
     max_tokens: chapterGenerationBudget.outputTokens,
   };
+  const deAiRequestOverrides: RequestOverrides = {
+    max_tokens: chapterDeAiBudget.outputTokens,
+  };
   // Same density as the token estimator / trimContextPack (CJK 1, English 4).
   const charsPerToken = charsPerTokenForLanguage();
   const totalContextCharBudget = totalContextTokenBudget * charsPerToken;
@@ -1384,7 +1398,7 @@ export async function runDeepChapterGeneration(
             signal,
             customDeAiSkill || undefined,
             cachePrefix,
-            generationRequestOverrides,
+            deAiRequestOverrides,
           ),
         (value) =>
           `简单审查与去AI味完成,最终正文约 ${countChapterChars(value)} 字。`,

+ 3 - 9
src/lib/novel/deep-chapter-prompts.spec.ts

@@ -1,7 +1,6 @@
 import { describe, expect, it } from "vitest"
 import {
   DEEP_CHAPTER_DRAFT_MAX_CHARS,
-  DEEP_CHAPTER_MAX_OUTPUT_TOKENS,
   DEEP_CHAPTER_MIN_CHARS,
   DEEP_CHAPTER_TARGET_CHARS,
   buildDeepChapterBriefPrompt,
@@ -16,22 +15,17 @@ describe("resolveChapterLengthSpec", () => {
     expect(spec.targetChars).toBe(DEEP_CHAPTER_TARGET_CHARS)
     expect(spec.minChars).toBe(DEEP_CHAPTER_MIN_CHARS)
     expect(spec.draftMaxChars).toBe(DEEP_CHAPTER_DRAFT_MAX_CHARS)
-    expect(spec.maxOutputTokens).toBe(DEEP_CHAPTER_MAX_OUTPUT_TOKENS)
+    expect(spec).not.toHaveProperty("maxOutputTokens")
   })
 
-  it("derives all thresholds from a configured chapter target (issue #8)", () => {
+  it("derives char thresholds from a configured chapter target (issue #8)", () => {
     const spec = resolveChapterLengthSpec(2000)
 
     expect(spec.targetChars).toBe(2000)
     expect(spec.minChars).toBeLessThan(2000)
     expect(spec.minChars).toBeGreaterThan(1000)
     expect(spec.draftMaxChars).toBe(2500)
-  })
-
-  it("scales output token budget up for long chapters", () => {
-    const spec = resolveChapterLengthSpec(6000)
-
-    expect(spec.maxOutputTokens).toBeGreaterThan(DEEP_CHAPTER_MAX_OUTPUT_TOKENS)
+    expect(spec).not.toHaveProperty("maxOutputTokens")
   })
 
   it("clamps unreasonable targets", () => {

+ 2 - 6
src/lib/novel/deep-chapter-prompts.ts

@@ -5,21 +5,19 @@ import { CHINESE_NOVEL_DE_AI_RULES } from "./de-ai-rules"
 export const DEEP_CHAPTER_TARGET_CHARS = 3000
 export const DEEP_CHAPTER_MIN_CHARS = 2200
 export const DEEP_CHAPTER_DRAFT_MAX_CHARS = 3500
-export const DEEP_CHAPTER_MAX_OUTPUT_TOKENS = 8000
 
-/** 章节生成字数规格:由设置中的“单章目标字数”推算(issue #8)。 */
+/** 章节生成字数规格:由设置中的“单章目标字数”推算(issue #8)。
+ *  输出 token 预算由 planChapterRequestBudget 按窗口比例规划,不再挂在此规格上。 */
 export interface ChapterLengthSpec {
   targetChars: number
   minChars: number
   draftMaxChars: number
-  maxOutputTokens: number
 }
 
 export const DEFAULT_CHAPTER_LENGTH_SPEC: ChapterLengthSpec = {
   targetChars: DEEP_CHAPTER_TARGET_CHARS,
   minChars: DEEP_CHAPTER_MIN_CHARS,
   draftMaxChars: DEEP_CHAPTER_DRAFT_MAX_CHARS,
-  maxOutputTokens: DEEP_CHAPTER_MAX_OUTPUT_TOKENS,
 }
 
 export function resolveChapterLengthSpec(targetChars?: number): ChapterLengthSpec {
@@ -32,8 +30,6 @@ export function resolveChapterLengthSpec(targetChars?: number): ChapterLengthSpe
     // 与默认 2200/3000 保持同一比例,最低不少于 300 字
     minChars: Math.max(300, Math.round(target * (DEEP_CHAPTER_MIN_CHARS / DEEP_CHAPTER_TARGET_CHARS))),
     draftMaxChars: target + (DEEP_CHAPTER_DRAFT_MAX_CHARS - DEEP_CHAPTER_TARGET_CHARS),
-    // 中文正文约 1-2 token/字,给草稿上限留足输出空间
-    maxOutputTokens: Math.max(DEEP_CHAPTER_MAX_OUTPUT_TOKENS, Math.ceil((target + 500) * 2)),
   }
 }
 

+ 22 - 1
src/lib/novel/deep-outline-generation.ts

@@ -2,6 +2,14 @@ import type { LlmConfig } from "@/stores/wiki-store"
 import { useWikiStore } from "@/stores/wiki-store"
 import { hasUsableLlm } from "@/lib/has-usable-llm"
 import { streamChat, type ChatMessage, type RequestOverrides, type StreamCallbacks } from "@/lib/llm-client"
+import {
+  planOutlineRequestBudget,
+  type OutlineBudgetStage,
+} from "@/lib/context-budget"
+import {
+  getEffectiveMaxOutputTokens,
+  thinkingMinMaxTokens,
+} from "@/lib/llm-providers"
 import { USER_ABORT_MESSAGE } from "@/lib/user-abort"
 
 export interface DeepOutlineGenerationInput {
@@ -69,6 +77,7 @@ export async function runDeepOutlineGeneration(
     deps,
     signal,
     (partial) => callbacks.onThinking?.(formatStageThinking("阶段2:大纲任务书", partial)),
+    "analysis",
   )
   callbacks.onThinking?.(formatStageThinking("阶段2:大纲任务书", taskBrief))
 
@@ -78,6 +87,7 @@ export async function runDeepOutlineGeneration(
     deps,
     signal,
     (partial) => callbacks.onThinking?.(formatStageThinking("阶段3:大纲草稿", partial)),
+    "generation",
   )
   callbacks.onThinking?.(formatStageThinking("阶段3:大纲草稿", [
     draftContent,
@@ -91,6 +101,7 @@ export async function runDeepOutlineGeneration(
     deps,
     signal,
     (partial) => callbacks.onThinking?.(formatStageThinking("阶段4:大纲自检", partial)),
+    "analysis",
   )
   callbacks.onThinking?.(formatStageThinking("阶段4:大纲自检", selfCheck))
   callbacks.onThinking?.(formatStageThinking("阶段5:完成", "采用自检后的大纲草稿作为最终输出。"))
@@ -110,9 +121,16 @@ async function collectModelText(
   deps: DeepOutlineGenerationDeps,
   signal?: AbortSignal,
   onUpdate?: (content: string) => void,
+  budgetStage: OutlineBudgetStage = "generation",
 ): Promise<string> {
   let content = ""
   let streamError: Error | null = null
+  const requestBudget = planOutlineRequestBudget({
+    maxContextSize: config.maxContextSize,
+    stage: budgetStage,
+    maxOutputTokens: getEffectiveMaxOutputTokens(config),
+    thinkingFloorTokens: thinkingMinMaxTokens(config.reasoning ?? { mode: "auto" }),
+  })
 
   await deps.streamChat(
     config,
@@ -128,7 +146,10 @@ async function collectModelText(
       },
     },
     signal,
-    { reasoning: config.reasoning },
+    {
+      reasoning: config.reasoning,
+      max_tokens: requestBudget.outputTokens,
+    },
   )
 
   if (signal?.aborted) throw new Error(USER_ABORT_MESSAGE)

+ 18 - 3
src/lib/novel/outline-generation.ts

@@ -1,10 +1,15 @@
 import { createDirectory, fileExists, listDirectory, readFile, writeFile } from "@/commands/fs"
 import { streamChat } from "@/lib/llm-client"
+import { planOutlineRequestBudget } from "@/lib/context-budget"
 import { getOutputLanguage } from "@/lib/output-language"
 import { getFileName, normalizePath } from "@/lib/path-utils"
 import { refreshProjectState } from "@/lib/project-refresh"
 import i18n from "@/i18n"
-import type { ChatMessage } from "@/lib/llm-providers"
+import {
+  getEffectiveMaxOutputTokens,
+  thinkingMinMaxTokens,
+  type ChatMessage,
+} from "@/lib/llm-providers"
 import { PROMPTS } from "@/lib/novel/prompt-templates"
 import { useOutlineGenerationStore } from "@/stores/outline-generation-store"
 import { useImportProgressStore } from "@/stores/import-progress-store"
@@ -254,6 +259,16 @@ function buildSectionRefinementPrompt(
   ].join("\n")
 }
 
+function outlineGenerationOverrides(llmConfig: LlmConfig) {
+  const budget = planOutlineRequestBudget({
+    maxContextSize: llmConfig.maxContextSize,
+    stage: "generation",
+    maxOutputTokens: getEffectiveMaxOutputTokens(llmConfig),
+    thinkingFloorTokens: thinkingMinMaxTokens(llmConfig.reasoning ?? { mode: "auto" }),
+  })
+  return { max_tokens: budget.outputTokens }
+}
+
 async function streamOutlineSectionContent(
   llmConfig: LlmConfig,
   context: string,
@@ -272,7 +287,7 @@ async function streamOutlineSectionContent(
     onError: (err) => {
       streamError = err
     },
-  }, signal)
+  }, signal, outlineGenerationOverrides(llmConfig))
 
   if (streamError) throw streamError
   return content.trim()
@@ -426,7 +441,7 @@ export async function generateOutlineFile(
     onError: (err) => {
       streamError = err
     },
-  }, signal)
+  }, signal, outlineGenerationOverrides(llmConfig))
 
   if (streamError) {
     throw streamError

+ 13 - 2
src/lib/novel/previous-chapters-analysis.ts

@@ -1,7 +1,11 @@
 import type { LlmConfig } from "@/stores/wiki-store"
 import { readFile } from "@/commands/fs"
 import { searchWiki } from "@/lib/search"
-import { computeContextBudget } from "@/lib/context-budget"
+import { computeContextBudget, planChapterRequestBudget } from "@/lib/context-budget"
+import {
+  getEffectiveMaxOutputTokens,
+  thinkingMinMaxTokens,
+} from "@/lib/llm-providers"
 
 /** 前情正文占分析模型窗口的比例;其余留给分析指令与模型输出。 */
 const PREVIOUS_BODY_WINDOW_FRAC = 0.5
@@ -76,8 +80,14 @@ export async function analyzePreviousChapters(
   // 构建分析prompt
   const analysisPrompt = buildPreviousChaptersAnalysisPrompt(budgetedChapters, currentChapterNumber)
 
-  // 调用LLM分析
+  // 调用LLM分析:显式挂 analysis 比例预算,避免只靠内核默认 generation 比例。
   const { streamChat } = await import("@/lib/llm-client")
+  const analysisBudget = planChapterRequestBudget({
+    maxContextSize: llmConfig.maxContextSize,
+    stage: "analysis",
+    maxOutputTokens: getEffectiveMaxOutputTokens(llmConfig),
+    thinkingFloorTokens: thinkingMinMaxTokens(llmConfig.reasoning ?? { mode: "auto" }),
+  })
   let analysis = ""
 
   await streamChat(
@@ -89,6 +99,7 @@ export async function analyzePreviousChapters(
       onError: () => {},
     },
     signal,
+    { max_tokens: analysisBudget.outputTokens },
   )
 
   if (signal?.aborted) throw new Error("已停止生成")