Просмотр исходного кода

fix(llm): 修复 thinking 模式 reasoning_content 全链路回传

补齐采集漏采、tool-loop 空串省略,以及大纲/采访历史回放缺口,避免 DeepSeek/Kimi 多轮 tool 调用 HTTP 400。

Co-authored-by: Cursor <cursoragent@cursor.com>
darknessomi 2 месяцев назад
Родитель
Сommit
f331f21de8

+ 6 - 3
src/components/chat/chat-panel.tsx

@@ -1406,7 +1406,7 @@ export function ChatPanel() {
         updateAgentAssistantMessage(assistantMessage.id, (message) => ({
           ...message,
           content: message.content || record?.finalText || "Agent未返回内容。",
-          ...(accumulatedReasoningContent ? { reasoning_content: accumulatedReasoningContent } : {}),
+          reasoning_content: accumulatedReasoningContent,
           agentToolCalls: settleRunningAgentToolCalls(record?.toolCalls.length ? record.toolCalls : message.agentToolCalls),
           agentStages: settleRunningAgentStages(message.agentStages, "done"),
           references: (() => {
@@ -1432,6 +1432,7 @@ export function ChatPanel() {
           content: message.content
             ? `${message.content}\n\n出错:${error.message}`
             : `出错:${error.message}`,
+          reasoning_content: accumulatedReasoningContent,
           agentToolCalls: settleRunningAgentToolCalls(message.agentToolCalls, "error"),
           agentStages: settleRunningAgentStages(message.agentStages, "error"),
           contextTrace: contextTrace || message.contextTrace,
@@ -1704,7 +1705,9 @@ export function ChatPanel() {
         ).map((message) => ({
           role: message.role,
           content: message.content,
-          ...(message.reasoning_content ? { reasoning_content: message.reasoning_content } : {}),
+          ...(message.reasoning_content !== undefined
+            ? { reasoning_content: message.reasoning_content }
+            : {}),
         } satisfies AgentMessage)),
         { role: "user", content: userContent },
       ]
@@ -1807,7 +1810,7 @@ export function ChatPanel() {
               updateAgentAssistantMessage(assistantMessage.id, (message) => ({
                 ...message,
                 content: finalContent,
-                ...(accumulatedReasoningContent ? { reasoning_content: accumulatedReasoningContent } : {}),
+                reasoning_content: accumulatedReasoningContent,
                 isAgentRunning: false,
               }))
             },

+ 22 - 2
src/components/sources/outline-chat-panel.tsx

@@ -1865,6 +1865,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
       let followUpGenerationPrompt: string | null = null;
       let contextHubResult: ContextHubResult | null = null;
       let providerUsage: LlmUsage | undefined;
+      let accumulatedReasoningContent = "";
 
       try {
         const contextHub = getContextHub(normalizePath(project.path));
@@ -2047,12 +2048,13 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
             streamToUser?: boolean;
             statusText?: string;
           } = {},
-        ): Promise<{ text: string; record: AgentRunRecord; error?: Error }> => {
+        ): Promise<{ text: string; record: AgentRunRecord; error?: Error; reasoning_content: string }> => {
           const { agentConfig, registry } = buildConfigForSkillNames(
             optionsForRun.skillNames,
             optionsForRun.disableWriteTools,
           );
           let runText = "";
+          let runReasoningContent = "";
           let agentError: Error | null = null;
           if (optionsForRun.statusText) {
             if (isCurrentRun()) setStreamingContent(capturedConvId, optionsForRun.statusText);
@@ -2075,6 +2077,10 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
                   }
                 }
               },
+              onReasoningToken: (chunk) => {
+                runReasoningContent += chunk;
+                accumulatedReasoningContent += chunk;
+              },
               onToolCall: () => {},
               onToolResult: () => {},
               onToolError: () => {},
@@ -2104,7 +2110,12 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
           const errMsg = agentError?.message ?? "";
           const isLengthTruncated = errMsg.includes("输出被截断") || errMsg.includes("最大输出 token");
           if (agentError && !isLengthTruncated) throw agentError;
-          return { text: runText || record.finalText, record, error: agentError ?? undefined };
+          return {
+            text: runText || record.finalText,
+            record,
+            error: agentError ?? undefined,
+            reasoning_content: runReasoningContent,
+          };
         };
 
         const runSingleAgentFallback = async () => {
@@ -2550,6 +2561,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
         updateOutlineAssistantMessage(convId, assistantId, (message) => ({
           ...message,
           content: finalContent,
+          reasoning_content: accumulatedReasoningContent,
           sources: finalSources,
           showThinkingProcess: historyPlan.showThinkingProcess,
           agentToolCalls: shouldShowToolProcess
@@ -2678,6 +2690,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
           updateOutlineAssistantMessage(convId, assistantId, (message) => ({
             ...message,
             content: partial,
+            reasoning_content: accumulatedReasoningContent,
             agentToolCalls: historyPlan.showToolProcessOnError
               ? settleRunningAgentToolCalls(
                   message.agentToolCalls?.length ? message.agentToolCalls : hiddenToolCalls,
@@ -2693,6 +2706,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
             updateOutlineAssistantMessage(convId, assistantId, (message) => ({
               ...message,
               content: `生成失败:${errorMsg}`,
+              reasoning_content: accumulatedReasoningContent,
               agentToolCalls: historyPlan.showToolProcessOnError
                 ? settleRunningAgentToolCalls(
                     message.agentToolCalls?.length ? message.agentToolCalls : hiddenToolCalls,
@@ -3357,6 +3371,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
           userMemorySessionKey: capturedConvId,
         };
         let agentError: Error | null = null;
+        let accumulatedReasoningContent = "";
         const record = await new AgentRunner().run(
           agentConfig,
           registry,
@@ -3380,6 +3395,9 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
                 );
               }
             },
+            onReasoningToken: (chunk) => {
+              accumulatedReasoningContent += chunk;
+            },
             onToolCall: () => {},
             onToolResult: () => {},
             onToolError: () => {},
@@ -3404,6 +3422,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
                 assistantId,
                 (message) => ({
                   ...message,
+                  reasoning_content: accumulatedReasoningContent,
                   agentToolCalls: settleRunningAgentToolCalls(
                     message.agentToolCalls,
                   ),
@@ -3456,6 +3475,7 @@ export function OutlineChatPanel({ onClose }: { onClose: () => void }) {
           (message) => ({
             ...message,
             content: finalContent,
+            reasoning_content: accumulatedReasoningContent,
             sources,
             agentToolCalls: settleRunningAgentToolCalls(record.toolCalls.length ? record.toolCalls : message.agentToolCalls),
             isAgentRunning: false,

+ 85 - 0
src/lib/agent/runner.spec.ts

@@ -88,6 +88,91 @@ describe("AgentRunner", () => {
     expect(callbacks.onError).not.toHaveBeenCalled()
   })
 
+  it("replays reasoning_content on tool-call assistant messages in the next round", async () => {
+    const tool: Tool = {
+      name: "read_chapter",
+      description: "read",
+      category: "read",
+      parameters: { name: { type: "string", description: "name" } },
+      execute: vi.fn().mockResolvedValue("Chapter content"),
+    }
+    registry.register(tool)
+
+    let callCount = 0
+    mockStreamChat.mockImplementation(async (_config: unknown, _msgs: unknown[], cb: StreamCallbacks) => {
+      callCount += 1
+      if (callCount === 1) {
+        cb.onReasoningToken?.("先读章节")
+        cb.onToolCallDelta?.({ index: 0, id: "call_reason_1", name: "read_chapter" })
+        cb.onToolCallDelta?.({ index: 0, arguments: '{"name":"ch1"}' })
+        cb.onDone()
+        return
+      }
+      cb.onToken("写完了")
+      cb.onDone()
+    })
+
+    const config: AgentConfig = {
+      maxRounds: 3,
+      tools: [tool],
+      systemPrompt: "You are helpful",
+      llmConfig: mockLlmConfig,
+    }
+    const result = await runner.run(
+      config,
+      registry,
+      [systemMsg, userMsg],
+      { onText: vi.fn(), onToolCall: vi.fn(), onToolResult: vi.fn(), onToolError: vi.fn(), onDone: vi.fn(), onError: vi.fn() },
+      undefined,
+    )
+
+    const round2Messages = mockStreamChat.mock.calls[1][1] as AgentMessage[]
+    const toolAssistant = round2Messages.find((message) => message.role === "assistant" && message.tool_calls?.length)
+    expect(toolAssistant?.reasoning_content).toBe("先读章节")
+    expect(result.finalText).toBe("写完了")
+  })
+
+  it("always attaches reasoning_content for tool-call assistants even when empty", async () => {
+    const tool: Tool = {
+      name: "read_chapter",
+      description: "read",
+      category: "read",
+      parameters: { name: { type: "string", description: "name" } },
+      execute: vi.fn().mockResolvedValue("Chapter content"),
+    }
+    registry.register(tool)
+
+    let callCount = 0
+    mockStreamChat.mockImplementation(async (_config: unknown, _msgs: unknown[], cb: StreamCallbacks) => {
+      callCount += 1
+      if (callCount === 1) {
+        cb.onToolCallDelta?.({ index: 0, id: "call_empty_reason", name: "read_chapter" })
+        cb.onToolCallDelta?.({ index: 0, arguments: '{"name":"ch1"}' })
+        cb.onDone()
+        return
+      }
+      cb.onToken("完成")
+      cb.onDone()
+    })
+
+    await runner.run(
+      {
+        maxRounds: 3,
+        tools: [tool],
+        systemPrompt: "You are helpful",
+        llmConfig: mockLlmConfig,
+      },
+      registry,
+      [systemMsg, userMsg],
+      { onText: vi.fn(), onToolCall: vi.fn(), onToolResult: vi.fn(), onToolError: vi.fn(), onDone: vi.fn(), onError: vi.fn() },
+      undefined,
+    )
+
+    const round2Messages = mockStreamChat.mock.calls[1][1] as AgentMessage[]
+    const toolAssistant = round2Messages.find((message) => message.role === "assistant" && message.tool_calls?.length)
+    expect(toolAssistant).toHaveProperty("reasoning_content", "")
+  })
+
   it("passes cacheable system content blocks through to the provider layer", async () => {
     mockStreamChat.mockImplementation(async (_config: unknown, _msgs: unknown[], cb: StreamCallbacks) => {
       cb.onToken("完成")

+ 4 - 2
src/lib/agent/runner.ts

@@ -262,12 +262,14 @@ export class AgentRunner {
         return record
       }
 
-      // Add assistant message with tool calls
+      // Add assistant message with tool calls.
+      // DeepSeek/Kimi thinking mode requires reasoning_content on every
+      // tool-call assistant message in subsequent rounds — even "".
       const assistantMsg: AgentMessage = {
         role: "assistant",
         content: roundText || "",
         tool_calls: toolCalls,
-        ...(roundReasoningContent ? { reasoning_content: roundReasoningContent } : {}),
+        reasoning_content: roundReasoningContent,
       }
       workingMessages.push(assistantMsg)
 

+ 8 - 4
src/lib/llm-client.ts

@@ -389,12 +389,14 @@ export async function streamChat(
           if (lineBuffer.trim()) {
             const trimmed = lineBuffer.trim()
             recordUsage(trimmed)
+            // Always harvest reasoning first: some gateways emit
+            // reasoning_content and tool_calls on the same SSE line.
+            reasoningCharsObserved += countReasoningCharsInLine(trimmed)
+            recordReasoning(trimmed)
             const toolDelta = parseToolCallDeltaFromLine(trimmed)
             if (toolDelta) {
               callbacks.onToolCallDelta?.(toolDelta)
             } else {
-              reasoningCharsObserved += countReasoningCharsInLine(trimmed)
-              recordReasoning(trimmed)
               const token = providerConfig.parseStream(trimmed)
               if (token !== null) recordToken(token)
             }
@@ -409,13 +411,15 @@ export async function streamChat(
           const trimmed = line.trim()
           if (!trimmed) continue
           recordUsage(trimmed)
+          // Always harvest reasoning first: some gateways emit
+          // reasoning_content and tool_calls on the same SSE line.
+          reasoningCharsObserved += countReasoningCharsInLine(trimmed)
+          recordReasoning(trimmed)
           const toolDelta = parseToolCallDeltaFromLine(trimmed)
           if (toolDelta) {
             callbacks.onToolCallDelta?.(toolDelta)
             continue
           }
-          reasoningCharsObserved += countReasoningCharsInLine(trimmed)
-          recordReasoning(trimmed)
           const token = providerConfig.parseStream(trimmed)
           if (token !== null) recordToken(token)
         }

+ 31 - 0
src/lib/llm-client.usage.spec.ts

@@ -70,6 +70,37 @@ describe("streamChat usage", () => {
     expect(onError).not.toHaveBeenCalled()
   })
 
+  it("同行 tool_calls 仍触发 onReasoningToken", async () => {
+    const encoder = new TextEncoder()
+    const body = new ReadableStream<Uint8Array>({
+      start(controller) {
+        controller.enqueue(encoder.encode([
+          'data: {"choices":[{"delta":{"reasoning_content":"需要读章","tool_calls":[{"index":0,"id":"call_1","function":{"name":"read_chapter","arguments":"{}"}}]}}]}',
+          "data: [DONE]",
+          "",
+        ].join("\n")))
+        controller.close()
+      },
+    })
+    mocks.fetch.mockResolvedValue(new Response(body, { status: 200 }))
+    const onReasoningToken = vi.fn()
+    const onToolCallDelta = vi.fn()
+
+    await streamChat(config, [{ role: "user", content: "写第一章" }], {
+      onToken: vi.fn(),
+      onReasoningToken,
+      onToolCallDelta,
+      onDone: vi.fn(),
+      onError: vi.fn(),
+    })
+
+    expect(onReasoningToken).toHaveBeenCalledWith("需要读章")
+    expect(onToolCallDelta).toHaveBeenCalledWith(expect.objectContaining({
+      id: "call_1",
+      name: "read_chapter",
+    }))
+  })
+
   it("发送前把总输入限制在模型窗口的 85%", async () => {
     mocks.fetch.mockResolvedValue(new Response([
       'data: {"choices":[{"delta":{"content":"完成"}}]}',

+ 19 - 0
src/lib/llm-providers.spec.ts

@@ -23,6 +23,25 @@ function requestBody(config: LlmConfig): Record<string, unknown> {
 }
 
 describe("llm provider reasoning options", () => {
+  it("replays assistant reasoning_content including empty string", () => {
+    const body = getProviderConfig(customConfig()).buildBody([
+      { role: "user", content: "写第一章" },
+      {
+        role: "assistant",
+        content: "",
+        tool_calls: [{
+          id: "call_1",
+          type: "function",
+          function: { name: "read_chapter", arguments: "{}" },
+        }],
+        reasoning_content: "",
+      },
+      { role: "tool", content: "章节内容", tool_call_id: "call_1", name: "read_chapter" },
+    ]) as { messages: Array<{ reasoning_content?: string }> }
+
+    expect(body.messages[1]?.reasoning_content).toBe("")
+  })
+
   it("sends reasoning_effort for explicit custom OpenAI-compatible reasoning mode", () => {
     const body = requestBody(customConfig({ reasoning: { mode: "high" } }))
 

+ 1 - 1
src/lib/llm-providers.ts

@@ -435,7 +435,7 @@ function buildOpenAiBody(
     ...(m.tool_calls ? { tool_calls: m.tool_calls } : {}),
     ...(m.tool_call_id ? { tool_call_id: m.tool_call_id } : {}),
     ...(m.name ? { name: m.name } : {}),
-    ...(m.reasoning_content ? { reasoning_content: m.reasoning_content } : {}),
+    ...(m.reasoning_content !== undefined ? { reasoning_content: m.reasoning_content } : {}),
   }))
   const body: Record<string, unknown> = { messages: translated, stream: true, ...stripWireAgnosticOverrides(overrides) }
   if (overrides?.tools && overrides.tools.length > 0) {

+ 12 - 0
src/lib/novel/novel-generation-request-package.spec.ts

@@ -35,6 +35,18 @@ describe("???????", () => {
     ])).toEqual([{ role: "user", content: "??????" }, { role: "assistant", content: "??" }])
   })
 
+  it("preserves assistant reasoning_content for model replay", () => {
+    expect(mapOutlineMessagesForModel([
+      { role: "user", content: "继续完善大纲" },
+      { role: "assistant", content: "已更新人物设定", reasoning_content: "先核对冲突" },
+      { role: "assistant", content: "已确认方向", reasoning_content: "" },
+    ])).toEqual([
+      { role: "user", content: "继续完善大纲" },
+      { role: "assistant", content: "已更新人物设定", reasoning_content: "先核对冲突" },
+      { role: "assistant", content: "已确认方向", reasoning_content: "" },
+    ])
+  })
+
   it("????????????", () => {
     const value = createNovelGenerationRequestPackage(request, "??????")
     const messages = [

+ 14 - 3
src/lib/novel/novel-generation-request-package.ts

@@ -21,6 +21,7 @@ type OutlineModelMessage = {
   content: string
   novelGenerationRequest?: NovelGenerationRequestPackage
   isAgentRunning?: boolean
+  reasoning_content?: string
 }
 
 type OutlineConversationLike = {
@@ -81,15 +82,25 @@ export function getOutlineMessageModelContent(message: {
     : message.content
 }
 
-export function mapOutlineMessagesForModel(messages: OutlineModelMessage[]): Array<{ role: "user" | "assistant"; content: string }> {
+export function mapOutlineMessagesForModel(messages: OutlineModelMessage[]): Array<{
+  role: "user" | "assistant"
+  content: string
+  reasoning_content?: string
+}> {
   return messages
     .filter((message) => message.content.trim() && !message.isAgentRunning)
-    .map((message) => ({ role: message.role, content: getOutlineMessageModelContent(message) }))
+    .map((message) => ({
+      role: message.role,
+      content: getOutlineMessageModelContent(message),
+      ...(message.reasoning_content !== undefined
+        ? { reasoning_content: message.reasoning_content }
+        : {}),
+    }))
 }
 
 export function buildOutlineRegenerationInput(messages: OutlineModelMessage[]): {
   request: string
-  history: Array<{ role: "user" | "assistant"; content: string }>
+  history: Array<{ role: "user" | "assistant"; content: string; reasoning_content?: string }>
   structuredGeneration: boolean
 } {
   const available = messages.filter((message) => message.content.trim() && !message.isAgentRunning)

+ 1 - 0
src/lib/novel/outline-context-reuse.ts

@@ -19,6 +19,7 @@ export type OutlineContextPressureLevel = "low" | "medium" | "high"
 export interface OutlineAgentHistoryMessage {
   role: "user" | "assistant" | "tool" | "system"
   content: string
+  reasoning_content?: string
 }
 
 export interface OutlineContextReuseInput {

+ 12 - 1
src/lib/novel/story-simulation/agent-interview.ts

@@ -196,7 +196,13 @@ export async function interviewAgent(
       if (msg.role === "user") {
         messages.push({ role: "user", content: msg.content })
       } else if (msg.role === "agent") {
-        messages.push({ role: "assistant", content: msg.content })
+        messages.push({
+          role: "assistant",
+          content: msg.content,
+          ...(msg.reasoning_content !== undefined
+            ? { reasoning_content: msg.reasoning_content }
+            : {}),
+        })
       }
     }
 
@@ -212,6 +218,7 @@ export async function interviewAgent(
 
     // 调用 LLM 流式回复
     let fullText = ""
+    let reasoningContent = ""
     let streamError: Error | null = null
 
     await streamChat(
@@ -222,6 +229,9 @@ export async function interviewAgent(
           fullText += token
           onToken?.(token)
         },
+        onReasoningToken: (token) => {
+          reasoningContent += token
+        },
         onDone: () => {},
         onError: (err) => {
           streamError = err
@@ -241,6 +251,7 @@ export async function interviewAgent(
       agentId: agent.characterId,
       agentName: agent.name,
       content: fullText,
+      reasoning_content: reasoningContent,
       timestamp: new Date().toISOString(),
     }
     session.messages.push(agentMsg)

+ 2 - 0
src/lib/novel/story-simulation/types.ts

@@ -206,6 +206,8 @@ export interface AgentChatMessage {
   agentName?: string
   content: string
   timestamp: string
+  /** Thinking-model chain-of-thought; must be replayed on subsequent turns. */
+  reasoning_content?: string
 }
 
 export interface AgentChatSession {

+ 19 - 0
src/lib/reasoning-detector.spec.ts

@@ -15,4 +15,23 @@ describe("reasoning detector", () => {
     expect(extractReasoningTextFromLine(line)).toEqual(["先确认用户意图"])
     expect(countReasoningCharsInLine(line)).toBe("先确认用户意图".length)
   })
+
+  it("extracts DeepSeek delta.reasoning_content", () => {
+    const line = 'data: {"choices":[{"delta":{"reasoning_content":"先读大纲再写"}}]}'
+
+    expect(extractReasoningTextFromLine(line)).toEqual(["先读大纲再写"])
+    expect(countReasoningCharsInLine(line)).toBe("先读大纲再写".length)
+  })
+
+  it("extracts message.reasoning_content from final non-delta chunks", () => {
+    const line = 'data: {"choices":[{"message":{"role":"assistant","content":"正文","reasoning_content":"整包思考"}}]}'
+
+    expect(extractReasoningTextFromLine(line)).toEqual(["整包思考"])
+  })
+
+  it("extracts reasoning_content even when the same line also has tool_calls", () => {
+    const line = 'data: {"choices":[{"delta":{"reasoning_content":"调用工具","tool_calls":[{"index":0,"id":"call_1","function":{"name":"read_chapter","arguments":"{}"}}]}}]}'
+
+    expect(extractReasoningTextFromLine(line)).toEqual(["调用工具"])
+  })
 })

+ 8 - 1
src/lib/reasoning-detector.ts

@@ -60,7 +60,10 @@ export function extractReasoningTextFromLine(rawLine: string): string[] {
 
   try {
     const parsed = JSON.parse(data) as {
-      choices?: Array<{ delta?: { reasoning_content?: string; reasoning?: string } }>
+      choices?: Array<{
+        delta?: { reasoning_content?: string; reasoning?: string }
+        message?: { reasoning_content?: string; reasoning?: string }
+      }>
       type?: string
       delta?: string | { type?: string; text?: string; thinking?: string }
       candidates?: Array<{
@@ -73,6 +76,10 @@ export function extractReasoningTextFromLine(rawLine: string): string[] {
       const delta = choice.delta
       if (typeof delta?.reasoning_content === "string") out.push(delta.reasoning_content)
       if (typeof delta?.reasoning === "string") out.push(delta.reasoning)
+      // Non-streaming / final chunk may put reasoning on message instead of delta.
+      const message = choice.message
+      if (typeof message?.reasoning_content === "string") out.push(message.reasoning_content)
+      if (typeof message?.reasoning === "string") out.push(message.reasoning)
     }
 
     if (

+ 1 - 1
src/stores/chat-store.ts

@@ -427,6 +427,6 @@ export function chatMessagesToLLM(messages: DisplayMessage[]): ChatMessage[] {
   return messages.map((m) => ({
     role: m.role,
     content: m.content,
-    ...(m.reasoning_content ? { reasoning_content: m.reasoning_content } : {}),
+    ...(m.reasoning_content !== undefined ? { reasoning_content: m.reasoning_content } : {}),
   }))
 }

+ 2 - 0
src/stores/outline-chat-store.ts

@@ -92,6 +92,8 @@ export interface OutlineChatMessage {
   nextStepRecommendation?: NextStepRecommendation | null
   novelGenerationRequest?: NovelGenerationRequestPackage
   contextHubSnapshot?: ContextHubSnapshotRef
+  /** Thinking-model chain-of-thought; must be replayed on subsequent turns. */
+  reasoning_content?: string
 }
 
 export interface OutlineChatConversation {