From b38f2ef45cd04cca6157ca6c70b94af908094a10 Mon Sep 17 00:00:00 2001 From: kartik-mem0 Date: Sat, 21 Mar 2026 19:04:42 +0530 Subject: [PATCH] fix(qdrant): deduplicate $or/$not keys and add icontains debug warning MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Memory._process_metadata_filters() renames OR→$or and NOT→$not, but effective_filters also retains the original OR/NOT keys from the deepcopy of input_filters. Without dedup, the same sub-conditions were evaluated twice in the Qdrant filter. Normalize the filter dict upfront so $or/$not/$and map to OR/NOT/AND and only the first occurrence is kept. Also replace the fragile value.get("contains") or value.get("icontains") pattern with explicit key lookup, and add a logger.debug for icontains explaining that Qdrant MatchText case sensitivity depends on full-text index configuration. Co-Authored-By: Claude Opus 4.6 (1M context) --- mem0/vector_stores/qdrant.py | 40 ++++++++++++++++++++---------- tests/vector_stores/test_qdrant.py | 21 +++++++++++----- 2 files changed, 42 insertions(+), 19 deletions(-) diff --git a/mem0/vector_stores/qdrant.py b/mem0/vector_stores/qdrant.py index 32f60e1f1..ad65837bc 100644 --- a/mem0/vector_stores/qdrant.py +++ b/mem0/vector_stores/qdrant.py @@ -189,9 +189,15 @@ class Qdrant(VectorStoreBase): elif "contains" in value or "icontains" in value: # MatchText: with a full-text index, tokenized matching (all words must appear). # Without a full-text index, exact substring match. - # Note: icontains maps to the same MatchText — case sensitivity depends on - # the full-text index tokenizer configuration, not on this operator name. - text = value.get("contains") or value.get("icontains") + op = "icontains" if "icontains" in value else "contains" + text = value[op] + if op == "icontains": + logger.debug( + "icontains on field '%s': Qdrant MatchText case sensitivity depends on " + "full-text index configuration. Without a full-text index this behaves " + "as a case-sensitive substring match (same as 'contains').", + key, + ) return FieldCondition(key=key, match=MatchText(text=text)) else: supported = {"eq", "ne", "gt", "gte", "lt", "lte", "in", "nin", "contains", "icontains"} @@ -217,33 +223,41 @@ class Qdrant(VectorStoreBase): if not filters: return None + # Normalize $or/$not/$and → OR/NOT/AND and deduplicate. + # Memory._process_metadata_filters() renames OR→$or and NOT→$not, + # but effective_filters retains the original OR/NOT keys from + # deepcopy(input_filters). Without dedup the same sub-conditions + # would be evaluated twice. + key_map = {"$or": "OR", "$not": "NOT", "$and": "AND"} + normalized = {} + for key, value in filters.items(): + norm_key = key_map.get(key, key) + if norm_key not in normalized: + normalized[norm_key] = value + must = [] should = [] must_not = [] - for key, value in filters.items(): - # Normalize $or/$not (injected by Memory._process_metadata_filters) - # to OR/NOT so they're handled uniformly. - normalized_key = {"$or": "OR", "$not": "NOT", "$and": "AND"}.get(key, key) - - if normalized_key in ("AND", "OR", "NOT"): + for key, value in normalized.items(): + if key in ("AND", "OR", "NOT"): if not isinstance(value, list): raise ValueError( - f"{normalized_key} filter value must be a list of filter dicts, " + f"{key} filter value must be a list of filter dicts, " f"got {type(value).__name__}" ) - if normalized_key == "AND": + if key == "AND": for sub in value: built = self._create_filter(sub) if built: must.append(built) - elif normalized_key == "OR": + elif key == "OR": for sub in value: built = self._create_filter(sub) if built: should.append(built) - elif normalized_key == "NOT": + elif key == "NOT": for sub in value: built = self._create_filter(sub) if built: diff --git a/tests/vector_stores/test_qdrant.py b/tests/vector_stores/test_qdrant.py index 5913a40af..7aceb64f7 100644 --- a/tests/vector_stores/test_qdrant.py +++ b/tests/vector_stores/test_qdrant.py @@ -698,8 +698,13 @@ class TestQdrantEnhancedFilters(unittest.TestCase): self.assertEqual(len(result.must_not), 1) def test_memory_search_or_shape(self): - """Simulate exact shape Memory.search() sends for OR filters.""" - # Memory.search() preserves original OR AND adds $or + """Simulate exact shape Memory.search() sends for OR filters. + + effective_filters keeps the original OR key (via deepcopy of + input_filters) and _process_metadata_filters adds $or with the + same content. _create_filter should deduplicate so only the + first occurrence (OR) is used — exactly 2 should entries. + """ filters = { "OR": [{"category": "programming"}, {"category": "data"}], "user_id": "test_user", @@ -707,12 +712,15 @@ class TestQdrantEnhancedFilters(unittest.TestCase): } result = self.qdrant._create_filter(filters) self.assertIsInstance(result, Filter) - # OR and $or both contribute to should — duplicates are harmless self.assertIsNotNone(result.should) - self.assertGreaterEqual(len(result.should), 2) + # Deduplicated: OR wins, $or is skipped — exactly 2 entries + self.assertEqual(len(result.should), 2) def test_memory_search_not_shape(self): - """Simulate exact shape Memory.search() sends for NOT filters.""" + """Simulate exact shape Memory.search() sends for NOT filters. + + Same deduplication as OR: NOT wins, $not is skipped. + """ filters = { "NOT": [{"category": "spam"}], "user_id": "test_user", @@ -721,7 +729,8 @@ class TestQdrantEnhancedFilters(unittest.TestCase): result = self.qdrant._create_filter(filters) self.assertIsInstance(result, Filter) self.assertIsNotNone(result.must_not) - self.assertGreaterEqual(len(result.must_not), 1) + # Deduplicated: NOT wins, $not is skipped — exactly 1 entry + self.assertEqual(len(result.must_not), 1) def tearDown(self): del self.qdrant