fix(qdrant): deduplicate $or/$not keys and add icontains debug warning
Memory._process_metadata_filters() renames OR→$or and NOT→$not, but
effective_filters also retains the original OR/NOT keys from the
deepcopy of input_filters. Without dedup, the same sub-conditions were
evaluated twice in the Qdrant filter.
Normalize the filter dict upfront so $or/$not/$and map to OR/NOT/AND
and only the first occurrence is kept. Also replace the fragile
value.get("contains") or value.get("icontains") pattern with explicit
key lookup, and add a logger.debug for icontains explaining that Qdrant
MatchText case sensitivity depends on full-text index configuration.
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -189,9 +189,15 @@ class Qdrant(VectorStoreBase):
|
||||
elif "contains" in value or "icontains" in value:
|
||||
# MatchText: with a full-text index, tokenized matching (all words must appear).
|
||||
# Without a full-text index, exact substring match.
|
||||
# Note: icontains maps to the same MatchText — case sensitivity depends on
|
||||
# the full-text index tokenizer configuration, not on this operator name.
|
||||
text = value.get("contains") or value.get("icontains")
|
||||
op = "icontains" if "icontains" in value else "contains"
|
||||
text = value[op]
|
||||
if op == "icontains":
|
||||
logger.debug(
|
||||
"icontains on field '%s': Qdrant MatchText case sensitivity depends on "
|
||||
"full-text index configuration. Without a full-text index this behaves "
|
||||
"as a case-sensitive substring match (same as 'contains').",
|
||||
key,
|
||||
)
|
||||
return FieldCondition(key=key, match=MatchText(text=text))
|
||||
else:
|
||||
supported = {"eq", "ne", "gt", "gte", "lt", "lte", "in", "nin", "contains", "icontains"}
|
||||
@@ -217,33 +223,41 @@ class Qdrant(VectorStoreBase):
|
||||
if not filters:
|
||||
return None
|
||||
|
||||
# Normalize $or/$not/$and → OR/NOT/AND and deduplicate.
|
||||
# Memory._process_metadata_filters() renames OR→$or and NOT→$not,
|
||||
# but effective_filters retains the original OR/NOT keys from
|
||||
# deepcopy(input_filters). Without dedup the same sub-conditions
|
||||
# would be evaluated twice.
|
||||
key_map = {"$or": "OR", "$not": "NOT", "$and": "AND"}
|
||||
normalized = {}
|
||||
for key, value in filters.items():
|
||||
norm_key = key_map.get(key, key)
|
||||
if norm_key not in normalized:
|
||||
normalized[norm_key] = value
|
||||
|
||||
must = []
|
||||
should = []
|
||||
must_not = []
|
||||
|
||||
for key, value in filters.items():
|
||||
# Normalize $or/$not (injected by Memory._process_metadata_filters)
|
||||
# to OR/NOT so they're handled uniformly.
|
||||
normalized_key = {"$or": "OR", "$not": "NOT", "$and": "AND"}.get(key, key)
|
||||
|
||||
if normalized_key in ("AND", "OR", "NOT"):
|
||||
for key, value in normalized.items():
|
||||
if key in ("AND", "OR", "NOT"):
|
||||
if not isinstance(value, list):
|
||||
raise ValueError(
|
||||
f"{normalized_key} filter value must be a list of filter dicts, "
|
||||
f"{key} filter value must be a list of filter dicts, "
|
||||
f"got {type(value).__name__}"
|
||||
)
|
||||
|
||||
if normalized_key == "AND":
|
||||
if key == "AND":
|
||||
for sub in value:
|
||||
built = self._create_filter(sub)
|
||||
if built:
|
||||
must.append(built)
|
||||
elif normalized_key == "OR":
|
||||
elif key == "OR":
|
||||
for sub in value:
|
||||
built = self._create_filter(sub)
|
||||
if built:
|
||||
should.append(built)
|
||||
elif normalized_key == "NOT":
|
||||
elif key == "NOT":
|
||||
for sub in value:
|
||||
built = self._create_filter(sub)
|
||||
if built:
|
||||
|
||||
@@ -698,8 +698,13 @@ class TestQdrantEnhancedFilters(unittest.TestCase):
|
||||
self.assertEqual(len(result.must_not), 1)
|
||||
|
||||
def test_memory_search_or_shape(self):
|
||||
"""Simulate exact shape Memory.search() sends for OR filters."""
|
||||
# Memory.search() preserves original OR AND adds $or
|
||||
"""Simulate exact shape Memory.search() sends for OR filters.
|
||||
|
||||
effective_filters keeps the original OR key (via deepcopy of
|
||||
input_filters) and _process_metadata_filters adds $or with the
|
||||
same content. _create_filter should deduplicate so only the
|
||||
first occurrence (OR) is used — exactly 2 should entries.
|
||||
"""
|
||||
filters = {
|
||||
"OR": [{"category": "programming"}, {"category": "data"}],
|
||||
"user_id": "test_user",
|
||||
@@ -707,12 +712,15 @@ class TestQdrantEnhancedFilters(unittest.TestCase):
|
||||
}
|
||||
result = self.qdrant._create_filter(filters)
|
||||
self.assertIsInstance(result, Filter)
|
||||
# OR and $or both contribute to should — duplicates are harmless
|
||||
self.assertIsNotNone(result.should)
|
||||
self.assertGreaterEqual(len(result.should), 2)
|
||||
# Deduplicated: OR wins, $or is skipped — exactly 2 entries
|
||||
self.assertEqual(len(result.should), 2)
|
||||
|
||||
def test_memory_search_not_shape(self):
|
||||
"""Simulate exact shape Memory.search() sends for NOT filters."""
|
||||
"""Simulate exact shape Memory.search() sends for NOT filters.
|
||||
|
||||
Same deduplication as OR: NOT wins, $not is skipped.
|
||||
"""
|
||||
filters = {
|
||||
"NOT": [{"category": "spam"}],
|
||||
"user_id": "test_user",
|
||||
@@ -721,7 +729,8 @@ class TestQdrantEnhancedFilters(unittest.TestCase):
|
||||
result = self.qdrant._create_filter(filters)
|
||||
self.assertIsInstance(result, Filter)
|
||||
self.assertIsNotNone(result.must_not)
|
||||
self.assertGreaterEqual(len(result.must_not), 1)
|
||||
# Deduplicated: NOT wins, $not is skipped — exactly 1 entry
|
||||
self.assertEqual(len(result.must_not), 1)
|
||||
|
||||
def tearDown(self):
|
||||
del self.qdrant
|
||||
|
||||
Reference in New Issue
Block a user