From 3ae1120403c1b9bc286a01a7f1c3a1ad35a223e9 Mon Sep 17 00:00:00 2001 From: anishamahuli Date: Thu, 5 Mar 2026 12:12:53 -0500 Subject: [PATCH] Add normalize_facts() utility for malformed LLM fact extraction output Port of TypeScript FactRetrievalSchema to Python. Normalizes facts that smaller LLMs return as {"fact": "..."} or {"text": "..."} objects back into plain strings before embedding. --- mem0/memory/utils.py | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/mem0/memory/utils.py b/mem0/memory/utils.py index 8c11705c8..6cf4a477b 100644 --- a/mem0/memory/utils.py +++ b/mem0/memory/utils.py @@ -2,9 +2,9 @@ import hashlib import re from mem0.configs.prompts import ( + AGENT_MEMORY_EXTRACTION_PROMPT, FACT_RETRIEVAL_PROMPT, USER_MEMORY_EXTRACTION_PROMPT, - AGENT_MEMORY_EXTRACTION_PROMPT, ) @@ -52,6 +52,25 @@ def format_entities(entities): return "\n".join(formatted_lines) +def normalize_facts(raw_facts): + """Normalize LLM-extracted facts to a list of strings. + + Smaller LLMs (e.g. llama3.1:8b) sometimes return facts as objects + like {"fact": "..."} or {"text": "..."} instead of plain strings. + This mirrors the TypeScript FactRetrievalSchema validation. + """ + normalized = [] + for item in raw_facts: + if isinstance(item, str): + fact = item + elif isinstance(item, dict): + fact = item.get("fact") or item.get("text") or str(item) + else: + fact = str(item) + if fact: + normalized.append(fact) + return normalized + def remove_code_blocks(content: str) -> str: """