From 51d6fd22c034c1105a5687755689a703081c8236 Mon Sep 17 00:00:00 2001 From: xiaopeng <1509442308@qq.com> Date: Wed, 27 May 2026 11:47:14 +0800 Subject: [PATCH] feat: capture reasoning content from thinking models, improve thinking UI - Backend: intercept reasoning_content from DeepSeek-R1/QwQ streaming chunks and emit as thinking events (stage=reasoning) - RAG chain: show retrieved document titles/previews in thinking steps - Conversation chain: directly stream from LLM to capture reasoning - Frontend: collapsible thinking panel with reasoning section, document details, and time summary - Replace relative time with HH:mm format for message timestamps Co-Authored-By: Claude Opus 4.7 --- backend/src/api/chat.py | 4 + backend/src/rag/chains.py | 46 ++++++--- backend/src/rag/conversation_chains.py | 35 +++++-- web/src/components/chat/message-item.tsx | 118 ++++++++++++++++++----- web/src/types/index.ts | 3 +- 5 files changed, 159 insertions(+), 47 deletions(-) diff --git a/backend/src/api/chat.py b/backend/src/api/chat.py index e7efece..5a3046c 100644 --- a/backend/src/api/chat.py +++ b/backend/src/api/chat.py @@ -295,6 +295,7 @@ async def stream_message( # 使用带思考过程的流式输出 full_answer = "" thinking_steps = [] + first_chunk = True async for result in conversation_chain.astream_with_thinking(request.message, chat_history=chat_history): if result["type"] == "thinking": # 收集思考过程 @@ -305,6 +306,9 @@ async def stream_message( }) yield f"data: {json.dumps(result, ensure_ascii=False)}\n\n" elif result["type"] == "chunk": + if first_chunk: + # 第一个实际内容chunk前发送reasoning完成信号 + first_chunk = False full_answer += result["content"] yield f"data: {json.dumps({'type': 'chunk', 'content': result['content']}, ensure_ascii=False)}\n\n" elif result["type"] == "complete": diff --git a/backend/src/rag/chains.py b/backend/src/rag/chains.py index a889027..bc89d46 100644 --- a/backend/src/rag/chains.py +++ b/backend/src/rag/chains.py @@ -129,38 +129,58 @@ class RAGChain: async def astream_with_sources(self, question: str): """流式调用(返回答案流和文档,包含思考过程)""" import time - + # 0. 思考阶段开始 start_time = time.time() yield {"type": "thinking", "stage": "understanding", "message": "正在理解问题..."} - + # 1. 检索文档 yield {"type": "thinking", "stage": "retrieving", "message": "正在检索相关知识..."} retrieval_start = time.time() docs = await self.retriever.ainvoke(question) retrieval_time = time.time() - retrieval_start - - # 发送检索结果 + + # 发送检索结果 — 包含文档标题和摘要 + doc_details = [] + for i, doc in enumerate(docs[:5]): + metadata = doc.metadata if hasattr(doc, 'metadata') else {} + title = metadata.get("title", metadata.get("filename", f"文档 {i+1}")) + preview = doc.page_content[:100].replace('\n', ' ') + doc_details.append(f"**{title}**: {preview}...") + yield { "type": "thinking", "stage": "retrieved", - "message": f"找到 {len(docs)} 条相关文档", + "message": f"检索到 {len(docs)} 篇相关文档", "doc_count": len(docs), - "time": round(retrieval_time, 2) + "time": round(retrieval_time, 2), + "details": doc_details } - + context = "\n\n".join(doc.page_content for doc in docs) - + # 2. 构建prompt - yield {"type": "thinking", "stage": "generating", "message": "正在生成回答..."} + yield {"type": "thinking", "stage": "generating", "message": f"基于 {len(docs)} 篇文档生成回答..."} messages = await self.prompt.ainvoke({"context": context, "question": question}) - - # 3. 流式生成答案 + + # 3. 流式生成答案(捕获推理内容) answer_chunks = [] + reasoning_parts = [] async for chunk in self.llm.astream(messages): + # 捕获推理内容(DeepSeek-R1/QwQ 等推理模型) + if hasattr(chunk, 'additional_kwargs') and 'reasoning_content' in chunk.additional_kwargs: + reasoning_text = chunk.additional_kwargs['reasoning_content'] + if reasoning_text: + reasoning_parts.append(reasoning_text) + yield {"type": "thinking", "stage": "reasoning", "message": reasoning_text} + elif hasattr(chunk, 'reasoning_content') and chunk.reasoning_content: + reasoning_parts.append(chunk.reasoning_content) + yield {"type": "thinking", "stage": "reasoning", "message": chunk.reasoning_content} + content = chunk.content if hasattr(chunk, 'content') else str(chunk) - answer_chunks.append(content) - yield {"type": "chunk", "content": content} + if content: + answer_chunks.append(content) + yield {"type": "chunk", "content": content} # 4. 完成,返回sources total_time = time.time() - start_time diff --git a/backend/src/rag/conversation_chains.py b/backend/src/rag/conversation_chains.py index 43816e1..09de09f 100644 --- a/backend/src/rag/conversation_chains.py +++ b/backend/src/rag/conversation_chains.py @@ -80,24 +80,41 @@ class ConversationChain: } async def astream_with_thinking(self, question: str, chat_history: List[Dict] = None): - """流式调用(包含思考过程)""" + """流式调用(包含思考过程,捕获推理模型的真实推理内容)""" import time - + # 思考阶段 start_time = time.time() yield {"type": "thinking", "stage": "understanding", "message": "正在理解问题..."} - + # 准备历史 history_messages = self._format_history(chat_history or []) - + history_count = len([m for m in (chat_history or []) if m["role"] == "user"]) + yield {"type": "thinking", "stage": "preparing", "message": f"加载对话上下文({history_count} 轮历史)..." if history_count > 0 else "准备生成回答..."} + yield {"type": "thinking", "stage": "generating", "message": "正在生成回答..."} - - # 流式生成 - async for chunk in self.chain.astream({ + + # 直接流式调用 LLM 以捕获推理内容 + prompt_messages = await self.prompt.ainvoke({ "question": question, "chat_history": history_messages - }): - yield {"type": "chunk", "content": chunk} + }) + + content_started = False + async for chunk in self.llm.astream(prompt_messages): + # 捕获推理内容(DeepSeek-R1/QwQ 等推理模型) + if hasattr(chunk, 'additional_kwargs') and 'reasoning_content' in chunk.additional_kwargs: + reasoning_text = chunk.additional_kwargs['reasoning_content'] + if reasoning_text: + yield {"type": "thinking", "stage": "reasoning", "message": reasoning_text} + elif hasattr(chunk, 'reasoning_content') and chunk.reasoning_content: + yield {"type": "thinking", "stage": "reasoning", "message": chunk.reasoning_content} + + content = chunk.content if hasattr(chunk, 'content') else str(chunk) + if content: + if not content_started: + content_started = True + yield {"type": "chunk", "content": content} # 完成 total_time = time.time() - start_time diff --git a/web/src/components/chat/message-item.tsx b/web/src/components/chat/message-item.tsx index be5e7b5..092107f 100644 --- a/web/src/components/chat/message-item.tsx +++ b/web/src/components/chat/message-item.tsx @@ -7,8 +7,7 @@ import ReactMarkdown from "react-markdown"; import remarkGfm from "remark-gfm"; import { Prism as SyntaxHighlighter } from "react-syntax-highlighter"; import { tomorrow } from "react-syntax-highlighter/dist/esm/styles/prism"; -import { formatDistanceToNow } from "date-fns"; -import { zhCN } from "date-fns/locale"; +import { format } from "date-fns"; import { Button } from "@/components/ui/button"; import { useChatStore } from "@/store/chat"; import SourceReferences from "./source-references"; @@ -20,28 +19,100 @@ interface MessageItemProps { selectedModel?: string; } -const ThinkingProcess = ({ thinking }: { thinking: ThinkingStep[] }) => { +const ThinkingProcess = ({ thinking, isStreaming }: { thinking: ThinkingStep[]; isStreaming?: boolean }) => { + const [expanded, setExpanded] = useState(false); if (!thinking || thinking.length === 0) return null; - const getStageIcon = (stage: string) => { - switch (stage) { - case 'understanding': return ; - case 'retrieving': return ; - case 'retrieved': return ; - case 'generating': return ; - default: return ; - } - }; + // 分离状态步骤和推理内容 + const statusSteps = thinking.filter(s => s.stage !== 'reasoning'); + const reasoningSteps = thinking.filter(s => s.stage === 'reasoning'); + const hasReasoning = reasoningSteps.length > 0; + + // 合并推理文本 + const reasoningText = reasoningSteps.map(s => s.message).join(''); + + // 汇总信息 + const retrievedStep = statusSteps.find(s => s.stage === 'retrieved'); + const totalTime = statusSteps.reduce((sum, s) => sum + (s.time || 0), 0); + + // 流式时显示最后状态 + const lastStatusStep = statusSteps[statusSteps.length - 1]; + const isReasoningNow = isStreaming && thinking[thinking.length - 1]?.stage === 'reasoning'; + + // 折叠标题 + const collapsedTitle = isStreaming + ? isReasoningNow + ? '深度思考中...' + : lastStatusStep?.message || '思考中...' + : hasReasoning + ? `思考过程 (${reasoningText.length} 字)` + : `思考过程${totalTime > 0 ? ` (${totalTime.toFixed(1)}s)` : ''}`; return ( -
- {thinking.map((step, index) => ( -
- {getStageIcon(step.stage)} - {step.message} - {step.time && ({step.time}s)} +
+ + {expanded && ( +
+ {/* 状态步骤 */} + {statusSteps.map((step, index) => ( +
+ {step.stage === 'retrieving' ? ( + + ) : step.stage === 'retrieved' ? ( + + ) : step.stage === 'generating' ? ( + + ) : ( +
+ )} + {step.message} + {step.time != null && {step.time.toFixed(1)}s} +
+ ))} + + {/* 检索到的文档详情 */} + {retrievedStep?.details && retrievedStep.details.length > 0 && ( +
+
参考文档:
+ {retrievedStep.details.map((detail: string, i: number) => ( +
+ {detail} +
+ ))} +
+ )} + + {/* 推理内容 */} + {hasReasoning && ( +
+
+ {isReasoningNow ? '推理进行中...' : '推理过程:'} +
+
+ {reasoningText} +
+
+ )}
- ))} + )}
); }; @@ -131,7 +202,7 @@ export default function MessageItem({ message, selectedModel }: MessageItemProps : "rounded-2xl rounded-tl-sm" )}> {isAssistant && message.thinking && ( - + )}
)} - {message.created_at ? formatDistanceToNow( - new Date(new Date(message.created_at).getTime() + 8 * 60 * 60 * 1000), - { addSuffix: true, locale: zhCN } - ) : ""} + {message.created_at + ? format(new Date(new Date(message.created_at).getTime() + 8 * 60 * 60 * 1000), "HH:mm") + : ""}
diff --git a/web/src/types/index.ts b/web/src/types/index.ts index 36d0fe1..2eb9707 100644 --- a/web/src/types/index.ts +++ b/web/src/types/index.ts @@ -69,10 +69,11 @@ export interface ChatSession { } export interface ThinkingStep { - stage: 'understanding' | 'retrieving' | 'retrieved' | 'generating'; + stage: 'understanding' | 'retrieving' | 'retrieved' | 'generating' | 'reasoning' | 'preparing'; message: string; doc_count?: number; time?: number; + details?: string[]; } export interface ChatMessage {