浏览代码

fix(writing): 修复章节工作流模型预算串扰

按各阶段实际模型窗口计算输出预算,避免小窗口辅助模型压缩任务书输出。
共享资料包继续采用各阶段上下文预算的最小值,并补充混合窗口回归测试。
darknessomi 4 周之前
父节点
当前提交
9f1d764e61
共有 3 个文件被更改,包括 89 次插入 和 37 次删除
  1. 6 0
      src/lib/context-budget.test.ts
  2. 70 18
      src/lib/novel/deep-chapter-generation.spec.ts
  3. 13 19
      src/lib/novel/deep-chapter-generation.ts

+ 6 - 0
src/lib/context-budget.test.ts

@@ -183,6 +183,12 @@ describe("chapter request budget", () => {
       stage: "analysis",
     }).outputTokens).toBe(40_000)
   })
+
+  it("uses the normalized default window when a stage omits maxContextSize", () => {
+    expect(planChapterRequestBudget({
+      stage: "analysis",
+    }).outputTokens).toBe(8_192)
+  })
 })
 
 describe("charsPerTokenForLanguage", () => {

+ 70 - 18
src/lib/novel/deep-chapter-generation.spec.ts

@@ -1348,16 +1348,23 @@ describe("runDeepChapterGeneration", () => {
   })
 
   it("binds analysis and generation budgets to deep chapter model calls", async () => {
+    const previousState = useWikiStore.getState()
     const deps = createDeps()
-    const overrides: Array<RequestOverrides | undefined> = []
+    const requests: Array<{ model: string; maxTokens?: number }> = []
+    let sharedContextTokenBudget: number | undefined
+    vi.mocked(deps.buildContextPack).mockResolvedValue({ ...contextPack, outline: "" })
+    vi.mocked(deps.contextPackToPrompt).mockImplementation((_pack, tokenBudget) => {
+      sharedContextTokenBudget = tokenBudget
+      return "上下文包内容"
+    })
     vi.mocked(deps.streamChat).mockImplementation(async (
-      _config: LlmConfig,
+      config: LlmConfig,
       messages: ChatMessage[],
       callbacks: StreamCallbacks,
       _signal,
       requestOverrides,
     ) => {
-      overrides.push(requestOverrides)
+      requests.push({ model: config.model, maxTokens: requestOverrides?.max_tokens })
       const prompt = messagesPromptText(messages)
       const content = prompt.includes("简单审查") || prompt.includes("去AI味")
         ? chapterText("最终无上限正文", 3000)
@@ -1370,23 +1377,68 @@ describe("runDeepChapterGeneration", () => {
       callbacks.onDone()
     })
 
-    await runDeepChapterGeneration(
-      {
-        projectPath: "E:/Novel",
-        userRequest: "生成第三章",
-        chapterNumber: 3,
-        // Auto reasoning carries no output floor, so the per-stage budgets show through.
-        llmConfig: { ...llmConfig, reasoning: { mode: "auto" } },
+    useWikiStore.setState({
+      defaultLlmModel: "workflow-provider/workflow-model",
+      providerConfigs: {
+        "workflow-provider": {
+          enabled: true,
+          apiKey: "workflow-key",
+          baseUrl: "https://workflow.example.test/v1",
+          maxContextSize: 1_000_000,
+          maxOutputTokens: 393_216,
+          reasoning: { mode: "auto" },
+          savedModels: [{ id: "workflow", name: "Workflow", model: "workflow-model", createdAt: 1 }],
+        },
+        "deai-provider": {
+          enabled: true,
+          apiKey: "deai-key",
+          baseUrl: "https://deai.example.test/v1",
+          maxContextSize: 204_800,
+          maxOutputTokens: 65_536,
+          reasoning: { mode: "auto" },
+          savedModels: [{ id: "deai", name: "De-AI", model: "de-ai-model", createdAt: 2 }],
+        },
       },
-      {},
-      deps,
-    )
+      novelConfig: {
+        ...previousState.novelConfig,
+        defaultLlmModel: "workflow-provider/workflow-model",
+        deAiModel: "deai-provider/de-ai-model",
+        deepPreviousChaptersAnalysis: false,
+      },
+    })
 
-    expect(overrides.length).toBeGreaterThan(0)
-    expect(overrides.every((item) => typeof item?.max_tokens === "number")).toBe(true)
-    // Shared window clamps to 204800: analysis 0.04 → 8192, generation 0.15 → 30720.
-    expect(overrides.some((item) => item?.max_tokens === 8_192)).toBe(true)
-    expect(overrides.some((item) => item?.max_tokens === 30_720)).toBe(true)
+    try {
+      await runDeepChapterGeneration(
+        {
+          projectPath: "E:/Novel",
+          userRequest: "生成第三章",
+          chapterNumber: 3,
+          // Auto reasoning carries no output floor, so the per-stage budgets show through.
+          llmConfig: {
+            ...llmConfig,
+            model: "writer-model",
+            maxContextSize: 262_144,
+            maxOutputTokens: 393_216,
+            reasoning: { mode: "auto" },
+          },
+        },
+        {},
+        deps,
+      )
+
+      expect(requests).toContainEqual({ model: "workflow-model", maxTokens: 40_000 })
+      expect(requests).toContainEqual({ model: "writer-model", maxTokens: 39_321 })
+      expect(requests).toContainEqual({ model: "de-ai-model", maxTokens: 30_720 })
+      // 共享资料包仍受最小的 204800-token 去 AI 味模型限制。
+      expect(sharedContextTokenBudget).toBe(133_120)
+    } finally {
+      useWikiStore.setState({
+        aiChatModel: previousState.aiChatModel,
+        defaultLlmModel: previousState.defaultLlmModel,
+        providerConfigs: previousState.providerConfigs,
+        novelConfig: previousState.novelConfig,
+      })
+    }
   })
 
   it("raises stage output to the floor the configured reasoning level needs", async () => {

+ 13 - 19
src/lib/novel/deep-chapter-generation.ts

@@ -693,23 +693,11 @@ export async function runDeepChapterGeneration(
   );
   throwIfAborted(signal);
 
-  // 任务书(workflowConfig)、初稿/返修(writingConfig)、去AI味(deAiConfig)复用同一份
-  // outlinePrompt + contextPrompt。这三个环节可能用不同模型、各有独立上下文窗口,
-  // 因此预算取三者窗口的最小值:确保任一环节都不必依赖 llm-client 末级无差别截断
-  // (末级截断按字符砍,会绕过 ContextPack 的字段优先级),并让各环节看到一致的上下文。
-  const sharedContextWindows = [
-    writingConfig.maxContextSize,
-    workflowConfig.maxContextSize,
-    deAiConfig.maxContextSize,
-  ].filter((size): size is number => typeof size === "number" && size > 0);
-  const sharedContextWindow = sharedContextWindows.length > 0
-    ? Math.min(...sharedContextWindows)
-    : input.llmConfig.maxContextSize;
-
-  // 大纲与其余上下文共用同一窗口预算:资料包按三阶段最小窗分配。
-  // 输出上限与思考地板按各阶段实际调用的模型重算,不再只读入口 llmConfig。
+  // 任务书(workflowConfig)、初稿/返修(writingConfig)、去AI味(deAiConfig)可能
+  // 使用不同模型。每个阶段的输出预算必须按实际调用模型的窗口与输出上限计算;
+  // 否则一个小窗口的辅助模型会错误压缩大窗口工作流模型的任务书输出。
   const chapterAnalysisBudget = planChapterRequestBudget({
-    maxContextSize: sharedContextWindow,
+    maxContextSize: workflowConfig.maxContextSize,
     chapterTargetChars: novelConfig.chapterTargetChars,
     stage: "analysis",
     maxOutputTokens: getEffectiveMaxOutputTokens(workflowConfig),
@@ -718,7 +706,7 @@ export async function runDeepChapterGeneration(
     ),
   });
   const chapterGenerationBudget = planChapterRequestBudget({
-    maxContextSize: sharedContextWindow,
+    maxContextSize: writingConfig.maxContextSize,
     chapterTargetChars: novelConfig.chapterTargetChars,
     stage: "generation",
     maxOutputTokens: getEffectiveMaxOutputTokens(writingConfig),
@@ -727,7 +715,7 @@ export async function runDeepChapterGeneration(
     ),
   });
   const chapterDeAiBudget = planChapterRequestBudget({
-    maxContextSize: sharedContextWindow,
+    maxContextSize: deAiConfig.maxContextSize,
     chapterTargetChars: novelConfig.chapterTargetChars,
     stage: "generation",
     maxOutputTokens: getEffectiveMaxOutputTokens(deAiConfig),
@@ -735,7 +723,13 @@ export async function runDeepChapterGeneration(
       deAiConfig.reasoning ?? { mode: "auto" },
     ),
   });
-  const totalContextTokenBudget = chapterGenerationBudget.contextTokenBudget;
+  // 三个阶段仍复用同一份 outlinePrompt + contextPrompt。共享输入取各阶段
+  // 实际可用上下文预算的最小值,确保任一模型都无需依赖 llm-client 末级截断。
+  const totalContextTokenBudget = Math.min(
+    chapterAnalysisBudget.contextTokenBudget,
+    chapterGenerationBudget.contextTokenBudget,
+    chapterDeAiBudget.contextTokenBudget,
+  );
   const analysisRequestOverrides: RequestOverrides = {
     max_tokens: chapterAnalysisBudget.outputTokens,
   };