import logging from langchain.chat_models import BaseChatModel from deerflow.config import get_app_config from deerflow.config.app_config import AppConfig from deerflow.reflection import resolve_class from deerflow.tracing import build_tracing_callbacks logger = logging.getLogger(__name__) def _deep_merge_dicts(base: dict | None, override: dict) -> dict: """Recursively merge two dictionaries without mutating the inputs.""" merged = dict(base or {}) for key, value in override.items(): if isinstance(value, dict) and isinstance(merged.get(key), dict): merged[key] = _deep_merge_dicts(merged[key], value) else: merged[key] = value return merged def _vllm_disable_chat_template_kwargs(chat_template_kwargs: dict) -> dict: """Build the disable payload for vLLM/Qwen chat template kwargs.""" disable_kwargs: dict[str, bool] = {} if "thinking" in chat_template_kwargs: disable_kwargs["thinking"] = False if "enable_thinking" in chat_template_kwargs: disable_kwargs["enable_thinking"] = False return disable_kwargs # OpenAI-compatible model classes whose constructor takes ``base_url`` (not ``api_base``) # and to which the OpenAI-specific defaults below apply. _OPENAI_COMPAT_USE_PATHS = ( "langchain_openai:ChatOpenAI", "deerflow.models.patched_openai:PatchedChatOpenAI", ) def _enable_stream_usage_by_default(model_use_path: str, model_settings_from_config: dict) -> None: """Enable stream usage for OpenAI-compatible models unless explicitly configured. LangChain only auto-enables ``stream_usage`` for OpenAI models when no custom base URL or client is configured. DeerFlow frequently uses OpenAI-compatible gateways, so token usage tracking would otherwise stay empty and the TokenUsageMiddleware would have nothing to log. """ if model_use_path not in _OPENAI_COMPAT_USE_PATHS: return if "stream_usage" in model_settings_from_config: return if "base_url" in model_settings_from_config or "openai_api_base" in model_settings_from_config: model_settings_from_config["stream_usage"] = True def _normalize_openai_base_url(model_use_path: str, model_settings_from_config: dict) -> None: """Map the common ``api_base`` alias to ``base_url`` for OpenAI-compatible clients. ``langchain_openai:ChatOpenAI`` (and the ``PatchedChatOpenAI`` subclass) accept the OpenAI endpoint override as ``base_url`` (with ``openai_api_base`` as a legacy alias). Several providers in ``config.example.yaml`` use ``api_base`` for *other* model classes, so users frequently copy ``api_base`` onto a ChatOpenAI model by mistake. Because ``ModelConfig`` is ``extra="allow"``, the bad key is not caught at config-load time — it is forwarded to the constructor, which does not reject it but transfers it into ``model_kwargs``; that is then spread into every ``Completions.create()`` call and rejected by the OpenAI SDK at *request* time with an opaque ``unexpected keyword argument 'api_base'`` error (and the endpoint override is silently dropped). Rename it here so the model works as the user intended. """ if model_use_path not in _OPENAI_COMPAT_USE_PATHS: return if "api_base" not in model_settings_from_config: return if "base_url" in model_settings_from_config or "openai_api_base" in model_settings_from_config: # Canonical key already present; drop the alias to avoid a duplicate-intent kwarg. model_settings_from_config.pop("api_base", None) logger.warning("Model config sets both an endpoint key (base_url/openai_api_base) and 'api_base'; using the former and ignoring 'api_base'.") return model_settings_from_config["base_url"] = model_settings_from_config.pop("api_base") logger.debug("Normalized model config key 'api_base' -> 'base_url' for OpenAI-compatible client.") def _warn_unknown_model_settings(model_use_path: str, model_class, model_name: str, model_settings_from_config: dict) -> None: """Warn about config keys the OpenAI client will silently divert into ``model_kwargs``. ``ModelConfig`` is ``extra="allow"``, so a typo'd key (e.g. ``maxx_tokens``) is not caught at config-load time. LangChain's OpenAI client does not reject an unknown constructor kwarg — it emits a ``UserWarning`` and transfers the key into ``model_kwargs``, which is then spread into every ``Completions.create()`` call and rejected by the OpenAI SDK at *request* time with an opaque ``unexpected keyword argument`` error that is very hard to trace back to a config typo. This turns that latent failure into an explicit, actionable log line at model-build time. It is **scoped to the OpenAI-compatible family** (``_OPENAI_COMPAT_USE_PATHS``) — that is where the ``model_kwargs`` divert-and-crash behavior occurs and where the known field/alias set is accurate. Other providers (e.g. ``ChatAnthropic``) route extra kwargs differently and would false-positive against this allow-list, so they are intentionally left alone. Best-effort and non-fatal: it only fires when the class exposes a pydantic ``model_fields`` schema, treats both field names and their aliases as valid, and allow-lists the standard passthrough kwargs the factory injects and the OpenAI client accepts. """ if model_use_path not in _OPENAI_COMPAT_USE_PATHS: return known = getattr(model_class, "model_fields", None) if not known: return valid_names = set(known.keys()) for field in known.values(): alias = getattr(field, "alias", None) if alias: valid_names.add(alias) # Standard kwargs the factory injects or the OpenAI client accepts beyond declared fields. valid_names |= { "model", "model_kwargs", "extra_body", "default_headers", "default_query", "stream_usage", "stream_chunk_timeout", "reasoning_effort", } unknown = sorted(k for k in model_settings_from_config if k not in valid_names) if unknown: logger.warning( "Model '%s' (%s): config key(s) %s are not recognized parameters of the model class and will be forwarded as-is; this may raise at request time. Check for typos (e.g. 'maxx_tokens' -> 'max_tokens').", model_name, getattr(model_class, "__name__", "?"), unknown, ) # Default chunk-gap budget for OpenAI-compatible streaming responses. # # langchain-openai raises ``StreamChunkTimeoutError`` after this many seconds # without receiving a chunk. Its own default is 60s, which is too aggressive for # reasoning models (DeepSeek-R1, Doubao-thinking, GPT-5) whose first chunk can # legitimately take 90~150s. We default to 240s so the streaming layer rarely # trips on long thinking pauses; the LLMErrorHandlingMiddleware still retries # (budget=2) if a real stall happens. Users can override per-model in config.yaml. _DEFAULT_STREAM_CHUNK_TIMEOUT_SECONDS: float = 240.0 def _apply_stream_chunk_timeout_default(model_use_path: str, model_settings_from_config: dict) -> None: """Inject a generous ``stream_chunk_timeout`` for OpenAI-compatible clients. The ``stream_chunk_timeout`` kwarg is specific to ``langchain_openai:ChatOpenAI`` and is rejected by other providers' constructors as an unexpected keyword argument. Behaviour: * OpenAI-compatible path: an explicit value in ``config.yaml`` is preserved. An explicit ``null`` is dropped upstream by ``model_dump(exclude_none=True)`` and therefore treated as "unset", so the default is injected. * Non-OpenAI path: drop the key so it is never forwarded to an incompatible constructor (which would raise ``TypeError: unexpected keyword argument``). """ if model_use_path not in _OPENAI_COMPAT_USE_PATHS: model_settings_from_config.pop("stream_chunk_timeout", None) return if "stream_chunk_timeout" in model_settings_from_config: return model_settings_from_config["stream_chunk_timeout"] = _DEFAULT_STREAM_CHUNK_TIMEOUT_SECONDS def create_chat_model(name: str | None = None, thinking_enabled: bool = False, *, app_config: AppConfig | None = None, attach_tracing: bool = True, **kwargs) -> BaseChatModel: """Create a chat model instance from the config. Args: name: The name of the model to create. If None, the first model in the config will be used. thinking_enabled: Enable the model's extended-thinking mode when supported. app_config: Explicit application config; falls back to the cached global if omitted. attach_tracing: When True (default), attach tracing callbacks (Langfuse, LangSmith) directly to the model instance. Standalone callers — anything that invokes the model outside a LangGraph run that already wires tracing at the invocation root (``MemoryUpdater``, ad-hoc utilities, etc.) — keep this default so the model-level callback still produces traces. Callers that already attach tracing at the graph root (``make_lead_agent``, the in-graph ``TitleMiddleware``) MUST pass ``attach_tracing=False``; otherwise the same LLM call emits duplicate spans (one rooted at the graph, one at the model) and ``session_id`` / ``user_id`` metadata never reach the trace because the model becomes a nested observation whose ``langfuse_*`` keys get stripped. Returns: A chat model instance. """ config = app_config or get_app_config() if name is None: name = config.models[0].name model_config = config.get_model_config(name) if model_config is None: raise ValueError(f"Model {name} not found in config") from None model_class = resolve_class(model_config.use, BaseChatModel) model_settings_from_config = model_config.model_dump( exclude_none=True, exclude={ "use", "name", "display_name", "description", "supports_thinking", "supports_reasoning_effort", "when_thinking_enabled", "when_thinking_disabled", "thinking", "supports_vision", # Presentation-only metadata (consumed by the console's cost # display) — must never reach the provider client, which would # forward unknown kwargs into the completion request payload. "pricing", }, ) # Compute effective when_thinking_enabled by merging in the `thinking` shortcut field. # The `thinking` shortcut is equivalent to setting when_thinking_enabled["thinking"]. has_thinking_settings = (model_config.when_thinking_enabled is not None) or (model_config.thinking is not None) effective_wte: dict = dict(model_config.when_thinking_enabled) if model_config.when_thinking_enabled else {} if model_config.thinking is not None: merged_thinking = {**(effective_wte.get("thinking") or {}), **model_config.thinking} effective_wte = {**effective_wte, "thinking": merged_thinking} if thinking_enabled and has_thinking_settings: if not model_config.supports_thinking: raise ValueError(f"Model {name} does not support thinking. Set `supports_thinking` to true in the `config.yaml` to enable thinking.") from None if effective_wte: model_settings_from_config.update(effective_wte) if not thinking_enabled: if model_config.when_thinking_disabled is not None: # User-provided disable settings take full precedence model_settings_from_config.update(model_config.when_thinking_disabled) elif has_thinking_settings and effective_wte.get("extra_body", {}).get("thinking", {}).get("type"): # OpenAI-compatible gateway: thinking is nested under extra_body model_settings_from_config["extra_body"] = _deep_merge_dicts( model_settings_from_config.get("extra_body"), {"thinking": {"type": "disabled"}}, ) model_settings_from_config["reasoning_effort"] = "minimal" elif has_thinking_settings and (disable_chat_template_kwargs := _vllm_disable_chat_template_kwargs(effective_wte.get("extra_body", {}).get("chat_template_kwargs") or {})): # vLLM uses chat template kwargs to switch thinking on/off. model_settings_from_config["extra_body"] = _deep_merge_dicts( model_settings_from_config.get("extra_body"), {"chat_template_kwargs": disable_chat_template_kwargs}, ) elif has_thinking_settings and effective_wte.get("thinking", {}).get("type"): # Native langchain_anthropic: thinking is a direct constructor parameter model_settings_from_config["thinking"] = {"type": "disabled"} if not model_config.supports_reasoning_effort: kwargs.pop("reasoning_effort", None) model_settings_from_config.pop("reasoning_effort", None) # Normalize the api_base -> base_url alias FIRST, so the downstream OpenAI-compatible # heuristics (stream_usage / stream_chunk_timeout) see the canonical endpoint key. _normalize_openai_base_url(model_config.use, model_settings_from_config) _enable_stream_usage_by_default(model_config.use, model_settings_from_config) _apply_stream_chunk_timeout_default(model_config.use, model_settings_from_config) # For Codex Responses API models: map thinking mode to reasoning_effort from deerflow.models.openai_codex_provider import CodexChatModel if issubclass(model_class, CodexChatModel): # The ChatGPT Codex endpoint currently rejects max_tokens/max_output_tokens. model_settings_from_config.pop("max_tokens", None) # Use explicit reasoning_effort from frontend if provided (low/medium/high) explicit_effort = kwargs.pop("reasoning_effort", None) if not thinking_enabled: model_settings_from_config["reasoning_effort"] = "none" elif explicit_effort and explicit_effort in ("low", "medium", "high", "xhigh"): model_settings_from_config["reasoning_effort"] = explicit_effort elif "reasoning_effort" not in model_settings_from_config: model_settings_from_config["reasoning_effort"] = "medium" # For MindIE models: enforce conservative retry defaults. # Timeout normalization is handled inside MindIEChatModel itself. if getattr(model_class, "__name__", "") == "MindIEChatModel": # Enforce max_retries constraint to prevent cascading timeouts. model_settings_from_config["max_retries"] = model_settings_from_config.get("max_retries", 1) # Ensure stream_usage is enabled so that token usage metadata is available # in streaming responses. LangChain's BaseChatOpenAI only defaults # stream_usage=True when no custom base_url/api_base is set, so models # hitting third-party endpoints (e.g. doubao, deepseek) silently lose # usage data. We default it to True unless explicitly configured. if "stream_usage" not in model_settings_from_config and "stream_usage" not in kwargs: if "stream_usage" in getattr(model_class, "model_fields", {}): model_settings_from_config["stream_usage"] = True _warn_unknown_model_settings(model_config.use, model_class, name, model_settings_from_config) model_instance = model_class(**kwargs, **model_settings_from_config) if attach_tracing: callbacks = build_tracing_callbacks() if callbacks: existing_callbacks = model_instance.callbacks or [] model_instance.callbacks = [*existing_callbacks, *callbacks] logger.debug(f"Tracing attached to model '{name}' with providers={len(callbacks)}") return model_instance