Files
mem0/tests/utils/test_lemmatization.py
T
Soumil Rathi a33c557a25 feat(oss): port v3 pipeline with hybrid search, entity extraction, and additive scoring
Replace the 2-LLM-call add pipeline with a single-pass additive extraction
using ADDITIVE_EXTRACTION_PROMPT. Memories now accumulate (ADD-only) instead
of being updated/deleted during extraction.

Search pipeline upgraded to hybrid scoring combining three signals:
- Semantic similarity (vector search)
- BM25 keyword matching (native per vector store, 15 stores supported)
- Entity boost (spaCy NER with entity collection linking)

Combined score = (semantic + bm25 + entity_boost) / max_possible, where
max_possible adapts based on which signals are active.

Key changes:
- Add spaCy-based lemmatization for BM25 keyword search
- Add spaCy-based entity extraction (PROPER, QUOTED, COMPOUND, NOUN types)
- Add entity store as second vector collection ({collection}_entities)
- Add native keyword_search() to 15 vector store adapters
- Add batch embedding support (embed_batch) for OpenAI and Azure OpenAI
- Add message persistence in SQLite (rolling window of 10 per scope)
- Add additive scoring with adaptive normalization
- Add observation_date parameter to add()
- Add custom_instructions config field
- Default search threshold changed to 0.1

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-03-31 14:09:37 -07:00

68 lines
2.3 KiB
Python

import pytest
@pytest.fixture(autouse=True)
def _ensure_spacy():
"""Skip tests if spaCy model is not available."""
try:
import spacy
spacy.load("en_core_web_sm")
except Exception:
pytest.skip("spaCy en_core_web_sm model not available")
class TestLemmatizeForBm25:
def test_basic_lemmatization(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("The cats are running quickly")
assert "cat" in result
assert "run" in result or "running" in result
# Stop words and punctuation should be removed
assert "the" not in result.split()
def test_verb_forms_normalized(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("she attended multiple meetings yesterday")
assert "attend" in result or "attended" in result
assert "meeting" in result # -ing form preserved alongside lemma
# "multiple" is kept (not a spaCy stop word)
def test_ing_preservation(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("attending the morning meeting")
tokens = result.split()
# Should have both the lemma and the -ing form
assert "attending" in tokens or "attend" in tokens
def test_empty_string(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("")
assert result == ""
def test_punctuation_removed(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("Hello, world! How are you?")
assert "," not in result
assert "!" not in result
assert "?" not in result
def test_lowercased(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("PYTHON Programming LANGUAGE")
for token in result.split():
assert token == token.lower()
def test_stop_words_removed(self):
from mem0.utils.lemmatization import lemmatize_for_bm25
result = lemmatize_for_bm25("this is a very simple test of the system")
tokens = result.split()
for stop in ["this", "is", "a", "very", "of", "the"]:
assert stop not in tokens