From dbac83218f45db4f2ae4e057ee38636c70a971e5 Mon Sep 17 00:00:00 2001 From: Rakhee Singh Date: Tue, 31 Mar 2026 13:25:45 +0530 Subject: [PATCH] fix(vllm): forward response_format to OpenAI-compatible API (#4608) Co-authored-by: rasingh5 --- mem0/llms/vllm.py | 2 ++ tests/llms/test_vllm.py | 44 +++++++++++++++++++++++++++++++++++++++++ 2 files changed, 46 insertions(+) diff --git a/mem0/llms/vllm.py b/mem0/llms/vllm.py index f7cbfbc58..ce2b61956 100644 --- a/mem0/llms/vllm.py +++ b/mem0/llms/vllm.py @@ -99,6 +99,8 @@ class VllmLLM(LLMBase): } ) + if response_format: + params["response_format"] = response_format if tools: params["tools"] = tools params["tool_choice"] = tool_choice diff --git a/tests/llms/test_vllm.py b/tests/llms/test_vllm.py index b0acf4631..25b31fc3e 100644 --- a/tests/llms/test_vllm.py +++ b/tests/llms/test_vllm.py @@ -88,6 +88,50 @@ def test_generate_response_with_tools(mock_vllm_client): +def test_generate_response_with_response_format(mock_vllm_client): + config = BaseLlmConfig(model="Qwen/Qwen2.5-32B-Instruct", temperature=0.7, max_tokens=100, top_p=1.0) + llm = VllmLLM(config) + messages = [ + {"role": "system", "content": "You are a memory extraction assistant."}, + {"role": "user", "content": "I like hiking on weekends."}, + ] + + mock_response = Mock() + mock_response.choices = [Mock(message=Mock(content='{"facts": ["User likes hiking on weekends"]}'))] + mock_vllm_client.chat.completions.create.return_value = mock_response + + response = llm.generate_response(messages, response_format={"type": "json_object"}) + + mock_vllm_client.chat.completions.create.assert_called_once_with( + model="Qwen/Qwen2.5-32B-Instruct", + messages=messages, + temperature=0.7, + max_tokens=100, + top_p=1.0, + response_format={"type": "json_object"}, + ) + assert response == '{"facts": ["User likes hiking on weekends"]}' + + +def test_generate_response_without_response_format(mock_vllm_client): + config = BaseLlmConfig(model="Qwen/Qwen2.5-32B-Instruct", temperature=0.7, max_tokens=100, top_p=1.0) + llm = VllmLLM(config) + messages = [ + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "Tell me a joke."}, + ] + + mock_response = Mock() + mock_response.choices = [Mock(message=Mock(content="Why did the chicken cross the road?"))] + mock_vllm_client.chat.completions.create.return_value = mock_response + + response = llm.generate_response(messages) + + call_kwargs = mock_vllm_client.chat.completions.create.call_args[1] + assert "response_format" not in call_kwargs + assert response == "Why did the chicken cross the road?" + + def create_mocked_memory(): """Create a fully mocked Memory instance for testing.""" with patch('mem0.utils.factory.LlmFactory.create') as mock_llm_factory, \