From 069ea0887c2eb1f3cfa15e0261d902b3adee3011 Mon Sep 17 00:00:00 2001 From: Bartok Date: Fri, 5 Jun 2026 08:12:56 -0500 Subject: [PATCH] fix(llms): add is_reasoning_model config override for versioned deployments (#5327) Co-authored-by: kartik-mem0 --- mem0/configs/llms/azure.py | 6 +++ mem0/configs/llms/base.py | 10 +++++ mem0/configs/llms/openai.py | 5 +++ mem0/llms/azure_openai.py | 1 + mem0/llms/base.py | 14 ++++++- mem0/llms/openai.py | 1 + tests/llms/test_azure_openai.py | 66 +++++++++++++++++++++++++++++++++ tests/llms/test_openai.py | 47 +++++++++++++++++++++++ 8 files changed, 148 insertions(+), 2 deletions(-) diff --git a/mem0/configs/llms/azure.py b/mem0/configs/llms/azure.py index 749e39c02..8b08e601d 100644 --- a/mem0/configs/llms/azure.py +++ b/mem0/configs/llms/azure.py @@ -23,6 +23,7 @@ class AzureOpenAIConfig(BaseLlmConfig): vision_details: Optional[str] = "auto", reasoning_effort: Optional[str] = None, http_client_proxies: Optional[dict] = None, + is_reasoning_model: Optional[bool] = None, # Azure OpenAI-specific parameters azure_kwargs: Optional[Dict[str, Any]] = None, ): @@ -40,6 +41,10 @@ class AzureOpenAIConfig(BaseLlmConfig): vision_details: Vision detail level, defaults to "auto" reasoning_effort: Effort level for reasoning models ("low", "medium", "high"), defaults to None http_client_proxies: HTTP client proxy settings, defaults to None + is_reasoning_model: Explicit override for reasoning-model detection. + None (default) uses the name-based heuristic. Set True to drop + max_tokens/temperature (e.g. for versioned Azure deployments like + "gpt-5.4-nano-2026-03-17"), or False to force standard params. azure_kwargs: Azure-specific configuration, defaults to None """ # Initialize base parameters @@ -54,6 +59,7 @@ class AzureOpenAIConfig(BaseLlmConfig): vision_details=vision_details, reasoning_effort=reasoning_effort, http_client_proxies=http_client_proxies, + is_reasoning_model=is_reasoning_model, ) # Azure OpenAI-specific parameters diff --git a/mem0/configs/llms/base.py b/mem0/configs/llms/base.py index 477479933..7d16543c3 100644 --- a/mem0/configs/llms/base.py +++ b/mem0/configs/llms/base.py @@ -25,6 +25,7 @@ class BaseLlmConfig(ABC): vision_details: Optional[str] = "auto", reasoning_effort: Optional[str] = None, http_client_proxies: Optional[Union[Dict, str]] = None, + is_reasoning_model: Optional[bool] = None, ): """ Initialize a base configuration class instance for the LLM. @@ -54,6 +55,14 @@ class BaseLlmConfig(ABC): Defaults to None (uses the model's default reasoning effort) http_client_proxies: Proxy settings for HTTP client. Can be a dict or string. Defaults to None + is_reasoning_model: Explicit override for reasoning-model detection. + When None (default), the model is classified automatically from its + name (preserving existing behavior). Set to True to force the + reasoning-model parameter set (drop max_tokens and temperature), + or False to force the standard parameter set. Useful for + deployments with custom/versioned model names (e.g. Azure + "gpt-5.4-nano-2026-03-17") that the name-based heuristic cannot + recognize. Defaults to None """ self.model = model self.temperature = temperature @@ -64,4 +73,5 @@ class BaseLlmConfig(ABC): self.enable_vision = enable_vision self.vision_details = vision_details self.reasoning_effort = reasoning_effort + self.is_reasoning_model = is_reasoning_model self.http_client = httpx.Client(proxies=http_client_proxies) if http_client_proxies else None diff --git a/mem0/configs/llms/openai.py b/mem0/configs/llms/openai.py index a65a1de83..94f8a11f0 100644 --- a/mem0/configs/llms/openai.py +++ b/mem0/configs/llms/openai.py @@ -22,6 +22,7 @@ class OpenAIConfig(BaseLlmConfig): vision_details: Optional[str] = "auto", reasoning_effort: Optional[str] = None, http_client_proxies: Optional[dict] = None, + is_reasoning_model: Optional[bool] = None, # OpenAI-specific parameters openai_base_url: Optional[str] = None, models: Optional[List[str]] = None, @@ -47,6 +48,9 @@ class OpenAIConfig(BaseLlmConfig): vision_details: Vision detail level, defaults to "auto" reasoning_effort: Effort level for reasoning models ("low", "medium", "high"), defaults to None http_client_proxies: HTTP client proxy settings, defaults to None + is_reasoning_model: Explicit override for reasoning-model detection. + None (default) uses the name-based heuristic. Set True to drop + max_tokens/temperature, or False to force standard params. openai_base_url: OpenAI API base URL, defaults to None models: List of models for OpenRouter, defaults to None route: OpenRouter route strategy, defaults to "fallback" @@ -72,6 +76,7 @@ class OpenAIConfig(BaseLlmConfig): vision_details=vision_details, reasoning_effort=reasoning_effort, http_client_proxies=http_client_proxies, + is_reasoning_model=is_reasoning_model, ) # OpenAI-specific parameters diff --git a/mem0/llms/azure_openai.py b/mem0/llms/azure_openai.py index 01a6d8e39..eb72683e7 100644 --- a/mem0/llms/azure_openai.py +++ b/mem0/llms/azure_openai.py @@ -33,6 +33,7 @@ class AzureOpenAILLM(LLMBase): vision_details=config.vision_details, reasoning_effort=getattr(config, 'reasoning_effort', None), http_client_proxies=config.http_client, + is_reasoning_model=getattr(config, 'is_reasoning_model', None), ) super().__init__(config) diff --git a/mem0/llms/base.py b/mem0/llms/base.py index 8812122a5..f867b0564 100644 --- a/mem0/llms/base.py +++ b/mem0/llms/base.py @@ -43,13 +43,23 @@ class LLMBase(ABC): def _is_reasoning_model(self, model: str) -> bool: """ Check if the model is a reasoning model or GPT-5 series that doesn't support certain parameters. - + + An explicit ``is_reasoning_model`` on the config takes precedence over the + name-based heuristic. This lets deployments with custom/versioned model + names (e.g. Azure ``gpt-5.4-nano-2026-03-17``) opt in or out without + relying on string matching. When the config value is ``None`` (default), + classification falls back to the name-based heuristic below. + Args: model: The model name to check - + Returns: bool: True if the model is a reasoning model or GPT-5 series """ + explicit = getattr(self.config, "is_reasoning_model", None) + if explicit is not None: + return explicit + reasoning_models = { "o1", "o1-preview", "o3-mini", "o3", "gpt-5", "gpt-5o", "gpt-5o-mini", "gpt-5o-micro", diff --git a/mem0/llms/openai.py b/mem0/llms/openai.py index 5be81687f..314bbf4d1 100644 --- a/mem0/llms/openai.py +++ b/mem0/llms/openai.py @@ -31,6 +31,7 @@ class OpenAILLM(LLMBase): vision_details=config.vision_details, reasoning_effort=getattr(config, 'reasoning_effort', None), http_client_proxies=config.http_client, + is_reasoning_model=getattr(config, 'is_reasoning_model', None), ) super().__init__(config) diff --git a/tests/llms/test_azure_openai.py b/tests/llms/test_azure_openai.py index 76faa6772..164495199 100644 --- a/tests/llms/test_azure_openai.py +++ b/tests/llms/test_azure_openai.py @@ -186,6 +186,72 @@ def test_azure_config_accepts_reasoning_effort(): assert config.model == "o3-mini" +def test_is_reasoning_model_override_forces_reasoning_path(mock_openai_client): + """Versioned Azure gpt-5.x deployments can opt in via is_reasoning_model=True. + + Regression test for https://github.com/mem0ai/mem0/issues/5296 — the + name-based heuristic does not recognize dated deployment names like + ``gpt-5.4-nano-2026-03-17``, so the call sent max_tokens and Azure replied + 400. The explicit override forces the reasoning-model parameter set, which + drops max_tokens (and temperature). + """ + config = AzureOpenAIConfig( + model="gpt-5.4-nano-2026-03-17", + temperature=TEMPERATURE, + max_tokens=MAX_TOKENS, + is_reasoning_model=True, + ) + llm = AzureOpenAILLM(config) + messages = [{"role": "user", "content": "I have oily skin."}] + + mock_response = Mock() + mock_response.choices = [Mock(message=Mock(content="ok"))] + mock_openai_client.chat.completions.create.return_value = mock_response + + llm.generate_response(messages) + + call_kwargs = mock_openai_client.chat.completions.create.call_args[1] + assert "max_tokens" not in call_kwargs + assert "temperature" not in call_kwargs + + +def test_is_reasoning_model_override_false_keeps_standard_params(mock_openai_client): + """is_reasoning_model=False forces the standard param set even for o-series names.""" + config = AzureOpenAIConfig( + model="o3-mini", + temperature=TEMPERATURE, + max_tokens=MAX_TOKENS, + top_p=TOP_P, + is_reasoning_model=False, + ) + llm = AzureOpenAILLM(config) + messages = [{"role": "user", "content": "Hello"}] + + mock_response = Mock() + mock_response.choices = [Mock(message=Mock(content="ok"))] + mock_openai_client.chat.completions.create.return_value = mock_response + + llm.generate_response(messages) + + call_kwargs = mock_openai_client.chat.completions.create.call_args[1] + assert call_kwargs["max_tokens"] == MAX_TOKENS + assert call_kwargs["temperature"] == TEMPERATURE + + +def test_is_reasoning_model_defaults_to_name_heuristic(mock_openai_client): + """When is_reasoning_model is None (default), classification stays name-based.""" + config = AzureOpenAIConfig( + model="gpt-5.4-nano-2026-03-17", + temperature=TEMPERATURE, + max_tokens=MAX_TOKENS, + top_p=TOP_P, + ) + llm = AzureOpenAILLM(config) + # Unrecognized versioned name -> heuristic says "not reasoning" (unchanged). + assert config.is_reasoning_model is None + assert llm._is_reasoning_model("gpt-5.4-nano-2026-03-17") is False + + @pytest.mark.parametrize( "default_headers", [None, {"Firstkey": "FirstVal", "SecondKey": "SecondVal"}], diff --git a/tests/llms/test_openai.py b/tests/llms/test_openai.py index 14e91cc75..2c408c099 100644 --- a/tests/llms/test_openai.py +++ b/tests/llms/test_openai.py @@ -328,6 +328,53 @@ def test_is_reasoning_model_classification(mock_openai_client): assert llm._is_reasoning_model("gpt-4.1-nano-2025-04-14") is False +def test_is_reasoning_model_explicit_override(mock_openai_client): + """Explicit is_reasoning_model overrides the name-based heuristic both ways. + + See https://github.com/mem0ai/mem0/issues/5296 — deployments with custom or + versioned names need to opt in/out without relying on string matching. + None (default) must preserve the existing heuristic. + """ + # Force True: a name the heuristic would reject is now reasoning. + config_true = OpenAIConfig(model="gpt-5.4-nano-2026-03-17", is_reasoning_model=True) + llm_true = OpenAILLM(config_true) + assert llm_true._is_reasoning_model("gpt-5.4-nano-2026-03-17") is True + + # Force False: an o-series name the heuristic would accept is now standard. + config_false = OpenAIConfig(model="o3-mini", is_reasoning_model=False) + llm_false = OpenAILLM(config_false) + assert llm_false._is_reasoning_model("o3-mini") is False + + # None (default) preserves the existing heuristic. + config_none = OpenAIConfig(model="gpt-4.1") + llm_none = OpenAILLM(config_none) + assert config_none.is_reasoning_model is None + assert llm_none._is_reasoning_model("o3-mini") is True + assert llm_none._is_reasoning_model("gpt-5.4-mini") is False + + +def test_is_reasoning_model_override_generates_correct_params(mock_openai_client): + """End-to-end: is_reasoning_model=True drops max_tokens/temperature from the actual API call.""" + config = OpenAIConfig( + model="gpt-5.4-nano-2026-03-17", + temperature=0.7, + max_tokens=100, + is_reasoning_model=True, + ) + llm = OpenAILLM(config) + messages = [{"role": "user", "content": "Hello"}] + + mock_response = Mock() + mock_response.choices = [Mock(message=Mock(content="ok"))] + mock_openai_client.chat.completions.create.return_value = mock_response + + llm.generate_response(messages) + + call_kwargs = mock_openai_client.chat.completions.create.call_args[1] + assert "max_tokens" not in call_kwargs + assert "temperature" not in call_kwargs + + def test_callback_with_tools(mock_openai_client): mock_callback = Mock() config = OpenAIConfig(model="gpt-4.1-nano-2025-04-14", response_callback=mock_callback)