feat(models): support reasoning_content streaming

2026-04-01 15:47:43 +08:00
parent 9561578a2a
commit 264183cec2
28 changed files with 495 additions and 109 deletions
--- a/api/app/core/models/base.py
+++ b/api/app/core/models/base.py
@@ -14,6 +14,7 @@ from pydantic import BaseModel, Field
 from app.core.error_codes import BizCode
 from app.core.exceptions import BusinessException
 from app.models.models_model import ModelProvider, ModelType
+from app.core.models.volcano_chat import VolcanoChatOpenAI

 T = TypeVar("T")

@@ -25,6 +26,9 @@ class RedBearModelConfig(BaseModel):
    api_key: str
    base_url: Optional[str] = None
    is_omni: bool = False  # 是否为 Omni 模型
+    deep_thinking: bool = False  # 是否启用深度思考模式
+    thinking_budget_tokens: Optional[int] = None  # 深度思考 token 预算
+    support_thinking: bool = False  # 模型是否支持 enable_thinking 参数（capability 含 thinking）
    # 请求超时时间（秒）- 默认120秒以支持复杂的LLM调用，可通过环境变量 LLM_TIMEOUT 配置
    timeout: float = Field(default_factory=lambda: float(os.getenv("LLM_TIMEOUT", "120.0")))
    # 最大重试次数 - 默认2次以避免过长等待，可通过环境变量 LLM_MAX_RETRIES 配置
@@ -44,7 +48,7 @@ class RedBearModelFactory:
        # 打印供应商信息用于调试
        from app.core.logging_config import get_business_logger
        logger = get_business_logger()
-        logger.debug(f"获取模型参数 - Provider: {provider}, Model: {config.model_name}, is_omni: {config.is_omni}")
+        logger.debug(f"获取模型参数 - Provider: {provider}, Model: {config.model_name}, is_omni: {config.is_omni}, deep_thinking: {config.deep_thinking}")

        # dashscope 的 omni 模型使用 OpenAI 兼容模式
        if provider == ModelProvider.DASHSCOPE and config.is_omni:
@@ -58,7 +62,7 @@ class RedBearModelFactory:
                write=60.0,
                pool=10.0,
            )
-            params = {
+            params: Dict[str, Any] = {
                "model": config.model_name,
                "base_url": config.base_url,
                "api_key": config.api_key,
@@ -67,8 +71,19 @@ class RedBearModelFactory:
                **config.extra_params
            }
            # 流式模式下启用 stream_usage 以获取 token 统计
-            if config.extra_params.get("streaming"):
+            is_streaming = bool(config.extra_params.get("streaming"))
+            if is_streaming:
                params["stream_usage"] = True
+            # 只有支持 thinking 的模型才传 enable_thinking
+            if config.support_thinking:
+                model_kwargs: Dict[str, Any] = config.extra_params.get("model_kwargs", {})
+                if is_streaming:
+                    model_kwargs["enable_thinking"] = config.deep_thinking
+                    if config.deep_thinking and config.thinking_budget_tokens:
+                        model_kwargs["thinking_budget"] = config.thinking_budget_tokens
+                else:
+                    model_kwargs["enable_thinking"] = False
+                params["model_kwargs"] = model_kwargs
            return params

        if provider in [ModelProvider.OPENAI, ModelProvider.XINFERENCE, ModelProvider.GPUSTACK, ModelProvider.OLLAMA, ModelProvider.VOLCANO]:
@@ -82,7 +97,7 @@ class RedBearModelFactory:
                write=60.0,  # 写入超时：60秒
                pool=10.0,  # 连接池超时：10秒
            )
-            params = {
+            params: Dict[str, Any] = {
                "model": config.model_name,
                "base_url": config.base_url,
                "api_key": config.api_key,
@@ -93,17 +108,44 @@ class RedBearModelFactory:
            # 流式模式下启用 stream_usage 以获取 token 统计
            if config.extra_params.get("streaming"):
                params["stream_usage"] = True
+            # 深度思考模式
+            is_streaming = bool(config.extra_params.get("streaming"))
+            if is_streaming:
+                if provider == ModelProvider.VOLCANO:
+                    # 火山引擎深度思考仅流式调用支持，非流式时不传 thinking 参数
+                    thinking_config: Dict[str, Any] = {
+                        "type": "enabled" if config.deep_thinking else "disabled"
+                    }
+                    if config.deep_thinking and config.thinking_budget_tokens:
+                        thinking_config["budget_tokens"] = config.thinking_budget_tokens
+                    params["extra_body"] = {"thinking": thinking_config}
+                else:
+                    # 始终显式传递 enable_thinking，不支持该参数的模型（如 DeepSeek-R1）会直接忽略
+                    model_kwargs: Dict[str, Any] = config.extra_params.get("model_kwargs", {})
+                    model_kwargs["enable_thinking"] = config.deep_thinking
+                    if config.deep_thinking and config.thinking_budget_tokens:
+                        model_kwargs["thinking_budget"] = config.thinking_budget_tokens
+                    params["model_kwargs"] = model_kwargs
            return params
        elif provider == ModelProvider.DASHSCOPE:
-            # DashScope (通义千问) 使用自己的参数格式
-            # 注意: DashScopeEmbeddings 不支持 timeout 和 base_url 参数
-            # 只支持: model, dashscope_api_key, max_retries, client
-            return {
+            params = {
                "model": config.model_name,
                "dashscope_api_key": config.api_key,
                "max_retries": config.max_retries,
                **config.extra_params
            }
+            # 只有支持 thinking 的模型才传 enable_thinking
+            if config.support_thinking:
+                is_streaming = bool(config.extra_params.get("streaming"))
+                model_kwargs: Dict[str, Any] = config.extra_params.get("model_kwargs", {})
+                if is_streaming:
+                    model_kwargs["enable_thinking"] = config.deep_thinking
+                    if config.deep_thinking and config.thinking_budget_tokens:
+                        model_kwargs["thinking_budget"] = config.thinking_budget_tokens
+                else:
+                    model_kwargs["enable_thinking"] = False
+                params["model_kwargs"] = model_kwargs
+            return params
        elif provider == ModelProvider.BEDROCK:
            # Bedrock 使用 AWS 凭证
            # api_key 格式: "access_key_id:secret_access_key" 或只是 access_key_id
@@ -142,6 +184,13 @@ class RedBearModelFactory:
            elif "region_name" not in params:
                params["region_name"] = "us-east-1"  # 默认区域

+            # 深度思考模式：Claude 3.7 Sonnet 等支持思考的模型
+            # 通过 additional_model_request_fields 传递 thinking 块，关闭时不传（Bedrock 无 disabled 选项）
+            if config.deep_thinking:
+                budget = config.thinking_budget_tokens or 10000
+                params["additional_model_request_fields"] = {
+                    "thinking": {"type": "enabled", "budget_tokens": budget}
+                }
            return params
        else:
            raise BusinessException(f"不支持的提供商: {provider}", code=BizCode.PROVIDER_NOT_SUPPORTED)
@@ -168,7 +217,9 @@ def get_provider_llm_class(config: RedBearModelConfig, type: ModelType = ModelTy
    # dashscope 的 omni 模型使用 OpenAI 兼容模式
    if provider == ModelProvider.DASHSCOPE and config.is_omni:
        return ChatOpenAI
-    if provider in [ModelProvider.OPENAI, ModelProvider.XINFERENCE, ModelProvider.GPUSTACK, ModelProvider.VOLCANO]:
+    if provider == ModelProvider.VOLCANO:
+        return VolcanoChatOpenAI
+    if provider in [ModelProvider.OPENAI, ModelProvider.XINFERENCE, ModelProvider.GPUSTACK]:
        if type == ModelType.LLM:
            return OpenAI
        elif type == ModelType.CHAT:
--- a/api/app/core/models/scripts/bedrock_models.yaml
+++ b/api/app/core/models/scripts/bedrock_models.yaml
@@ -11,6 +11,7 @@ models:
  tags:
  - 大语言模型
  logo: bedrock
+
 - name: amazon nova
  type: llm
  provider: bedrock
@@ -27,6 +28,7 @@ models:
  - stream-tool-call
  - vision
  logo: bedrock
+
 - name: anthropic claude
  type: llm
  provider: bedrock
@@ -35,6 +37,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -44,6 +47,7 @@ models:
  - stream-tool-call
  - document
  logo: bedrock
+
 - name: cohere
  type: llm
  provider: bedrock
@@ -58,6 +62,7 @@ models:
  - tool-call
  - stream-tool-call
  logo: bedrock
+
 - name: deepseek
  type: llm
  provider: bedrock
@@ -66,6 +71,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -74,6 +80,7 @@ models:
  - tool-call
  - stream-tool-call
  logo: bedrock
+
 - name: meta
  type: llm
  provider: bedrock
@@ -87,6 +94,7 @@ models:
  - agent-thought
  - tool-call
  logo: bedrock
+
 - name: mistral
  type: llm
  provider: bedrock
@@ -100,6 +108,7 @@ models:
  - agent-thought
  - tool-call
  logo: bedrock
+
 - name: openai
  type: llm
  provider: bedrock
@@ -114,6 +123,7 @@ models:
  - tool-call
  - stream-tool-call
  logo: bedrock
+
 - name: qwen
  type: llm
  provider: bedrock
@@ -128,6 +138,7 @@ models:
  - tool-call
  - stream-tool-call
  logo: bedrock
+
 - name: amazon.rerank-v1:0
  type: rerank
  provider: bedrock
@@ -139,6 +150,7 @@ models:
  tags:
  - 重排序模型
  logo: bedrock
+
 - name: cohere.rerank-v3-5:0
  type: rerank
  provider: bedrock
@@ -150,6 +162,7 @@ models:
  tags:
  - 重排序模型
  logo: bedrock
+
 - name: amazon.nova-2-multimodal-embeddings-v1:0
  type: embedding
  provider: bedrock
@@ -163,6 +176,7 @@ models:
  - 文本嵌入模型
  - vision
  logo: bedrock
+
 - name: amazon.titan-embed-text-v1
  type: embedding
  provider: bedrock
@@ -174,6 +188,7 @@ models:
  tags:
  - 文本嵌入模型
  logo: bedrock
+
 - name: amazon.titan-embed-text-v2:0
  type: embedding
  provider: bedrock
@@ -185,6 +200,7 @@ models:
  tags:
  - 文本嵌入模型
  logo: bedrock
+
 - name: cohere.embed-english-v3
  type: embedding
  provider: bedrock
@@ -196,6 +212,7 @@ models:
  tags:
  - 文本嵌入模型
  logo: bedrock
+
 - name: cohere.embed-multilingual-v3
  type: embedding
  provider: bedrock
--- a/api/app/core/models/scripts/dashscope_models.yaml
+++ b/api/app/core/models/scripts/dashscope_models.yaml
@@ -6,36 +6,42 @@ models:
  description: DeepSeek-R1-Distill-Qwen-14B大语言模型，支持智能体思考，32000上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: deepseek-r1-distill-qwen-32b
  type: llm
  provider: dashscope
  description: DeepSeek-R1-Distill-Qwen-32B大语言模型，支持智能体思考，32000上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: deepseek-r1
  type: llm
  provider: dashscope
  description: DeepSeek-R1大语言模型，支持智能体思考，131072超大上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: deepseek-v3.1
  type: llm
  provider: dashscope
@@ -48,6 +54,7 @@ models:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: deepseek-v3.2-exp
  type: llm
  provider: dashscope
@@ -60,6 +67,7 @@ models:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: deepseek-v3.2
  type: llm
  provider: dashscope
@@ -72,6 +80,7 @@ models:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: deepseek-v3
  type: llm
  provider: dashscope
@@ -84,6 +93,7 @@ models:
  - 大语言模型
  - agent-thought
  logo: dashscope
+
 - name: farui-plus
  type: llm
  provider: dashscope
@@ -98,6 +108,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: glm-4.7
  type: llm
  provider: dashscope
@@ -112,6 +123,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qvq-max-latest
  type: llm
  provider: dashscope
@@ -119,7 +131,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
+  - vision
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -127,6 +140,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qvq-max
  type: llm
  provider: dashscope
@@ -134,7 +148,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
+  - vision
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -142,6 +157,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-coder-turbo-0919
  type: llm
  provider: dashscope
@@ -155,13 +171,15 @@ models:
  - 代码模型
  - agent-thought
  logo: dashscope
+
 - name: qwen-max-latest
  type: llm
  provider: dashscope
  description: qwen-max-latest大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式，支持联网搜索
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -169,6 +187,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-max-longcontext
  type: llm
  provider: dashscope
@@ -183,13 +202,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-max
  type: llm
  provider: dashscope
  description: qwen-max大语言模型，支持多工具调用、智能体思考、流式工具调用，32768上下文窗口，对话模式，支持联网搜索
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -197,6 +218,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-mt-plus
  type: llm
  provider: dashscope
@@ -210,6 +232,7 @@ models:
  - 翻译模型
  - agent-thought
  logo: dashscope
+
 - name: qwen-mt-turbo
  type: llm
  provider: dashscope
@@ -223,6 +246,7 @@ models:
  - 翻译模型
  - agent-thought
  logo: dashscope
+
 - name: qwen-plus-0112
  type: llm
  provider: dashscope
@@ -237,6 +261,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-0125
  type: llm
  provider: dashscope
@@ -251,6 +276,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-0723
  type: llm
  provider: dashscope
@@ -265,6 +291,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-0806
  type: llm
  provider: dashscope
@@ -279,6 +306,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-0919
  type: llm
  provider: dashscope
@@ -293,6 +321,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-1125
  type: llm
  provider: dashscope
@@ -307,6 +336,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-1127
  type: llm
  provider: dashscope
@@ -321,6 +351,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-plus-1220
  type: llm
  provider: dashscope
@@ -335,6 +366,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen-vl-max
  type: chat
  provider: dashscope
@@ -342,8 +374,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
  is_omni: false
  tags:
  - 大语言模型
@@ -352,6 +384,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen-vl-plus-0809
  type: chat
  provider: dashscope
@@ -359,8 +392,8 @@ models:
  is_deprecated: true
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
  is_omni: false
  tags:
  - 大语言模型
@@ -369,6 +402,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen-vl-plus-2025-01-02
  type: chat
  provider: dashscope
@@ -376,8 +410,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
  is_omni: false
  tags:
  - 大语言模型
@@ -386,6 +420,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen-vl-plus-2025-01-25
  type: chat
  provider: dashscope
@@ -393,8 +428,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
  is_omni: false
  tags:
  - 大语言模型
@@ -403,6 +438,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen-vl-plus-latest
  type: chat
  provider: dashscope
@@ -410,8 +446,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
  is_omni: false
  tags:
  - 大语言模型
@@ -420,6 +456,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen-vl-plus
  type: chat
  provider: dashscope
@@ -427,8 +464,8 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
  is_omni: false
  tags:
  - 大语言模型
@@ -437,6 +474,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen2.5-0.5b-instruct
  type: llm
  provider: dashscope
@@ -451,13 +489,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-14b
  type: llm
  provider: dashscope
  description: qwen3-14b大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -465,13 +505,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-235b-a22b-instruct-2507
  type: llm
  provider: dashscope
  description: qwen3-235b-a22b-instruct-2507大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -479,13 +521,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-235b-a22b-thinking-2507
  type: llm
  provider: dashscope
  description: qwen3-235b-a22b-thinking-2507大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -493,13 +537,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-235b-a22b
  type: llm
  provider: dashscope
  description: qwen3-235b-a22b大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -507,13 +553,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-30b-a3b-instruct-2507
  type: llm
  provider: dashscope
  description: qwen3-30b-a3b-instruct-2507大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -521,13 +569,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-30b-a3b
  type: llm
  provider: dashscope
  description: qwen3-30b-a3b大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -535,13 +585,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-32b
  type: llm
  provider: dashscope
  description: qwen3-32b大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -549,13 +601,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-4b
  type: llm
  provider: dashscope
  description: qwen3-4b大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -563,13 +617,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-8b
  type: llm
  provider: dashscope
  description: qwen3-8b大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -577,65 +633,75 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-coder-30b-a3b-instruct
  type: llm
  provider: dashscope
  description: qwen3-coder-30b-a3b-instruct大语言模型，支持智能体思考，262144上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - 代码模型
  - agent-thought
  logo: dashscope
+
 - name: qwen3-coder-480b-a35b-instruct
  type: llm
  provider: dashscope
  description: qwen3-coder-480b-a35b-instruct大语言模型，支持智能体思考，262144上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - 代码模型
  - agent-thought
  logo: dashscope
+
 - name: qwen3-coder-plus-2025-09-23
  type: llm
  provider: dashscope
  description: qwen3-coder-plus-2025-09-23大语言模型，支持智能体思考，1000000上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - 代码模型
  - agent-thought
  logo: dashscope
+
 - name: qwen3-coder-plus
  type: llm
  provider: dashscope
  description: qwen3-coder-plus大语言模型，支持智能体思考，1000000上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - 代码模型
  - agent-thought
  logo: dashscope
+
 - name: qwen3-max-2025-09-23
  type: llm
  provider: dashscope
  description: qwen3-max-2025-09-23大语言模型，支持多工具调用、智能体思考、流式工具调用，262144上下文窗口，对话模式，支持联网搜索
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -644,13 +710,15 @@ models:
  - stream-tool-call
  - 联网搜索
  logo: dashscope
+
 - name: qwen3-max-2026-01-23
  type: llm
  provider: dashscope
  description: qwen3-max-2026-01-23大语言模型，支持多工具调用、智能体思考、流式工具调用，262144上下文窗口，对话模式，支持联网搜索
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -659,13 +727,15 @@ models:
  - stream-tool-call
  - 联网搜索
  logo: dashscope
+
 - name: qwen3-max-preview
  type: llm
  provider: dashscope
  description: qwen3-max-preview大语言模型，支持多工具调用、智能体思考、流式工具调用，262144上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -673,13 +743,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-max
  type: llm
  provider: dashscope
  description: qwen3-max大语言模型，支持多工具调用、智能体思考、流式工具调用，262144上下文窗口，对话模式，支持联网搜索
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -688,13 +760,15 @@ models:
  - stream-tool-call
  - 联网搜索
  logo: dashscope
+
 - name: qwen3-next-80b-a3b-instruct
  type: llm
  provider: dashscope
  description: qwen3-next-80b-a3b-instruct大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -702,13 +776,15 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-next-80b-a3b-thinking
  type: llm
  provider: dashscope
  description: qwen3-next-80b-a3b-thinking大语言模型，支持多工具调用、智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -716,6 +792,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwen3-omni-flash-2025-12-01
  type: llm
  provider: dashscope
@@ -723,9 +800,10 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
-    - audio
+  - vision
+  - video
+  - audio
+  - thinking
  is_omni: true
  tags:
  - 大语言模型
@@ -735,6 +813,7 @@ models:
  - video
  - audio
  logo: dashscope
+
 - name: qwen3-vl-235b-a22b-instruct
  type: chat
  provider: dashscope
@@ -742,8 +821,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -754,6 +834,7 @@ models:
  - vision
  - video
  logo: dashscope
+
 - name: qwen3-vl-235b-a22b-thinking
  type: chat
  provider: dashscope
@@ -761,8 +842,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -773,6 +855,7 @@ models:
  - vision
  - video
  logo: dashscope
+
 - name: qwen3-vl-30b-a3b-instruct
  type: chat
  provider: dashscope
@@ -780,8 +863,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -792,6 +876,7 @@ models:
  - vision
  - video
  logo: dashscope
+
 - name: qwen3-vl-30b-a3b-thinking
  type: chat
  provider: dashscope
@@ -799,8 +884,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -811,6 +897,7 @@ models:
  - vision
  - video
  logo: dashscope
+
 - name: qwen3-vl-flash
  type: chat
  provider: dashscope
@@ -818,8 +905,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -830,6 +918,7 @@ models:
  - vision
  - video
  logo: dashscope
+
 - name: qwen3-vl-plus-2025-09-23
  type: chat
  provider: dashscope
@@ -837,8 +926,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -847,6 +937,7 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwen3-vl-plus
  type: chat
  provider: dashscope
@@ -854,8 +945,9 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
-    - video
+  - vision
+  - video
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -864,45 +956,52 @@ models:
  - agent-thought
  - video
  logo: dashscope
+
 - name: qwq-32b
  type: llm
  provider: dashscope
  description: qwq-32b大语言模型，支持智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwq-plus-0305
  type: llm
  provider: dashscope
  description: qwq-plus-0305大语言模型，支持智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: qwq-plus
  type: llm
  provider: dashscope
  description: qwq-plus大语言模型，支持智能体思考、流式工具调用，131072上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+  - thinking
  is_omni: false
  tags:
  - 大语言模型
  - agent-thought
  - stream-tool-call
  logo: dashscope
+
 - name: gte-rerank-v2
  type: rerank
  provider: dashscope
@@ -914,6 +1013,7 @@ models:
  tags:
  - 重排序模型
  logo: dashscope
+
 - name: gte-rerank
  type: rerank
  provider: dashscope
@@ -925,6 +1025,7 @@ models:
  tags:
  - 重排序模型
  logo: dashscope
+
 - name: multimodal-embedding-v1
  type: embedding
  provider: dashscope
@@ -932,13 +1033,14 @@ models:
  is_deprecated: false
  is_official: true
  capability:
-    - vision
+  - vision
  is_omni: false
  tags:
  - 嵌入模型
  - 多模态模型
  - vision
  logo: dashscope
+
 - name: text-embedding-v1
  type: embedding
  provider: dashscope
@@ -951,6 +1053,7 @@ models:
  - 嵌入模型
  - 文本嵌入
  logo: dashscope
+
 - name: text-embedding-v2
  type: embedding
  provider: dashscope
@@ -963,6 +1066,7 @@ models:
  - 嵌入模型
  - 文本嵌入
  logo: dashscope
+
 - name: text-embedding-v3
  type: embedding
  provider: dashscope
@@ -975,6 +1079,7 @@ models:
  - 嵌入模型
  - 文本嵌入
  logo: dashscope
+
 - name: text-embedding-v4
  type: embedding
  provider: dashscope
@@ -986,4 +1091,4 @@ models:
  tags:
  - 嵌入模型
  - 文本嵌入
-  logo: dashscope
+  logo: dashscope
--- a/api/app/core/models/scripts/openai_models.yaml
+++ b/api/app/core/models/scripts/openai_models.yaml
@@ -20,6 +20,7 @@ models:
  - audio
  - video
  logo: openai
+
 - name: gpt-3.5-turbo-0125
  type: llm
  provider: openai
@@ -34,6 +35,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-3.5-turbo-1106
  type: llm
  provider: openai
@@ -48,6 +50,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-3.5-turbo-16k
  type: llm
  provider: openai
@@ -62,6 +65,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-3.5-turbo-instruct
  type: llm
  provider: openai
@@ -73,6 +77,7 @@ models:
  tags:
  - 大语言模型
  logo: openai
+
 - name: gpt-3.5-turbo
  type: llm
  provider: openai
@@ -87,6 +92,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-4-0125-preview
  type: llm
  provider: openai
@@ -101,6 +107,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-4-1106-preview
  type: llm
  provider: openai
@@ -115,6 +122,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-4-turbo-2024-04-09
  type: llm
  provider: openai
@@ -131,6 +139,7 @@ models:
  - stream-tool-call
  - vision
  logo: openai
+
 - name: gpt-4-turbo-preview
  type: llm
  provider: openai
@@ -145,6 +154,7 @@ models:
  - agent-thought
  - stream-tool-call
  logo: openai
+
 - name: gpt-4-turbo
  type: llm
  provider: openai
@@ -161,6 +171,7 @@ models:
  - stream-tool-call
  - vision
  logo: openai
+
 - name: o1-preview
  type: llm
  provider: openai
@@ -173,6 +184,7 @@ models:
  - 大语言模型
  - agent-thought
  logo: openai
+
 - name: o1
  type: llm
  provider: openai
@@ -181,6 +193,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -190,6 +203,7 @@ models:
  - vision
  - structured-output
  logo: openai
+
 - name: o3-2025-04-16
  type: llm
  provider: openai
@@ -198,6 +212,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -207,13 +222,15 @@ models:
  - stream-tool-call
  - structured-output
  logo: openai
+
 - name: o3-mini-2025-01-31
  type: llm
  provider: openai
  description: o3-mini-2025-01-31大语言模型，支持智能体思考、工具调用、流式工具调用、结构化输出，200000上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -222,13 +239,15 @@ models:
  - stream-tool-call
  - structured-output
  logo: openai
+
 - name: o3-mini
  type: llm
  provider: openai
  description: o3-mini大语言模型，支持智能体思考、工具调用、流式工具调用、结构化输出，200000上下文窗口，对话模式
  is_deprecated: false
  is_official: true
-  capability: []
+  capability:
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -237,6 +256,7 @@ models:
  - stream-tool-call
  - structured-output
  logo: openai
+
 - name: o3-pro-2025-06-10
  type: llm
  provider: openai
@@ -245,6 +265,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -253,6 +274,7 @@ models:
  - vision
  - structured-output
  logo: openai
+
 - name: o3-pro
  type: llm
  provider: openai
@@ -261,6 +283,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -269,6 +292,7 @@ models:
  - vision
  - structured-output
  logo: openai
+
 - name: o3
  type: llm
  provider: openai
@@ -277,6 +301,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -286,6 +311,7 @@ models:
  - stream-tool-call
  - structured-output
  logo: openai
+
 - name: o4-mini-2025-04-16
  type: llm
  provider: openai
@@ -294,6 +320,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -303,6 +330,7 @@ models:
  - stream-tool-call
  - structured-output
  logo: openai
+
 - name: o4-mini
  type: llm
  provider: openai
@@ -311,6 +339,7 @@ models:
  is_official: true
  capability:
    - vision
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -320,6 +349,7 @@ models:
  - stream-tool-call
  - structured-output
  logo: openai
+
 - name: text-embedding-3-large
  type: embedding
  provider: openai
@@ -331,6 +361,7 @@ models:
  tags:
  - 文本向量模型
  logo: openai
+
 - name: text-embedding-3-small
  type: embedding
  provider: openai
@@ -342,6 +373,7 @@ models:
  tags:
  - 文本向量模型
  logo: openai
+
 - name: text-embedding-ada-002
  type: embedding
  provider: openai
--- a/api/app/core/models/scripts/volcano_models.yaml
+++ b/api/app/core/models/scripts/volcano_models.yaml
@@ -10,6 +10,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -24,6 +25,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -38,6 +40,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -52,6 +55,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -82,6 +86,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -96,6 +101,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -110,6 +116,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -124,6 +131,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
@@ -139,6 +147,7 @@ models:
  capability:
    - vision
    - video
+    - thinking
  is_omni: false
  tags:
  - 大语言模型
--- a/api/app/core/models/volcano_chat.py
+++ b/api/app/core/models/volcano_chat.py
@@ -0,0 +1,38 @@
+"""
+火山引擎 ChatOpenAI 扩展
+
+ChatOpenAI 在解析流式 SSE 时只取 delta.content，会丢弃 delta.reasoning_content。
+此类仅重写 _convert_chunk_to_generation_chunk，将 reasoning_content 补入 additional_kwargs。
+"""
+from __future__ import annotations
+
+from typing import Any, Optional
+
+from langchain_core.outputs import ChatGenerationChunk
+from langchain_openai import ChatOpenAI
+
+
+class VolcanoChatOpenAI(ChatOpenAI):
+    """火山引擎 Chat 模型，支持深度思考内容（reasoning_content）的流式透传。"""
+
+    def _convert_chunk_to_generation_chunk(
+        self,
+        chunk: dict,
+        default_chunk_class: type,
+        base_generation_info: Optional[dict],
+    ) -> Optional[ChatGenerationChunk]:
+        gen_chunk = super()._convert_chunk_to_generation_chunk(
+            chunk, default_chunk_class, base_generation_info
+        )
+        if gen_chunk is None:
+            return None
+
+        # 从原始 chunk 中提取 reasoning_content
+        choices = chunk.get("choices") or chunk.get("chunk", {}).get("choices", [])
+        if choices:
+            delta = choices[0].get("delta") or {}
+            reasoning: Any = delta.get("reasoning_content")
+            if reasoning:
+                gen_chunk.message.additional_kwargs["reasoning_content"] = reasoning
+
+        return gen_chunk