a33c557a25
Replace the 2-LLM-call add pipeline with a single-pass additive extraction
using ADDITIVE_EXTRACTION_PROMPT. Memories now accumulate (ADD-only) instead
of being updated/deleted during extraction.
Search pipeline upgraded to hybrid scoring combining three signals:
- Semantic similarity (vector search)
- BM25 keyword matching (native per vector store, 15 stores supported)
- Entity boost (spaCy NER with entity collection linking)
Combined score = (semantic + bm25 + entity_boost) / max_possible, where
max_possible adapts based on which signals are active.
Key changes:
- Add spaCy-based lemmatization for BM25 keyword search
- Add spaCy-based entity extraction (PROPER, QUOTED, COMPOUND, NOUN types)
- Add entity store as second vector collection ({collection}_entities)
- Add native keyword_search() to 15 vector store adapters
- Add batch embedding support (embed_batch) for OpenAI and Azure OpenAI
- Add message persistence in SQLite (rolling window of 10 per scope)
- Add additive scoring with adaptive normalization
- Add observation_date parameter to add()
- Add custom_instructions config field
- Default search threshold changed to 0.1
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
68 lines
2.3 KiB
Python
68 lines
2.3 KiB
Python
import pytest
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _ensure_spacy():
|
|
"""Skip tests if spaCy model is not available."""
|
|
try:
|
|
import spacy
|
|
spacy.load("en_core_web_sm")
|
|
except Exception:
|
|
pytest.skip("spaCy en_core_web_sm model not available")
|
|
|
|
|
|
class TestLemmatizeForBm25:
|
|
def test_basic_lemmatization(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("The cats are running quickly")
|
|
assert "cat" in result
|
|
assert "run" in result or "running" in result
|
|
# Stop words and punctuation should be removed
|
|
assert "the" not in result.split()
|
|
|
|
def test_verb_forms_normalized(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("she attended multiple meetings yesterday")
|
|
assert "attend" in result or "attended" in result
|
|
assert "meeting" in result # -ing form preserved alongside lemma
|
|
# "multiple" is kept (not a spaCy stop word)
|
|
|
|
def test_ing_preservation(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("attending the morning meeting")
|
|
tokens = result.split()
|
|
# Should have both the lemma and the -ing form
|
|
assert "attending" in tokens or "attend" in tokens
|
|
|
|
def test_empty_string(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("")
|
|
assert result == ""
|
|
|
|
def test_punctuation_removed(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("Hello, world! How are you?")
|
|
assert "," not in result
|
|
assert "!" not in result
|
|
assert "?" not in result
|
|
|
|
def test_lowercased(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("PYTHON Programming LANGUAGE")
|
|
for token in result.split():
|
|
assert token == token.lower()
|
|
|
|
def test_stop_words_removed(self):
|
|
from mem0.utils.lemmatization import lemmatize_for_bm25
|
|
|
|
result = lemmatize_for_bm25("this is a very simple test of the system")
|
|
tokens = result.split()
|
|
for stop in ["this", "is", "a", "very", "of", "the"]:
|
|
assert stop not in tokens
|