From 4c41f6deeb5b804d4467ad1f2570612f55134ac6 Mon Sep 17 00:00:00 2001 From: Eldar Shlomi <72104254+eldar702@users.noreply.github.com> Date: Thu, 11 Jun 2026 18:58:42 +0300 Subject: [PATCH] fix(vector-stores): index Valkey 'memory' field as TEXT not TAG (#5443) Co-authored-by: Claude Opus 4.8 Co-authored-by: kartik-mem0 --- mem0/vector_stores/valkey.py | 4 +-- tests/vector_stores/test_valkey.py | 41 ++++++++++++++++++++++++++++++ 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/mem0/vector_stores/valkey.py b/mem0/vector_stores/valkey.py index 4a4c80dde..7ba333d19 100644 --- a/mem0/vector_stores/valkey.py +++ b/mem0/vector_stores/valkey.py @@ -21,7 +21,7 @@ DEFAULT_FIELDS = [ {"name": "agent_id", "type": "tag"}, {"name": "run_id", "type": "tag"}, {"name": "user_id", "type": "tag"}, - {"name": "memory", "type": "tag"}, # Using TAG instead of TEXT for Valkey compatibility + {"name": "memory", "type": "text"}, # TEXT for full-text search over memory content (see #5006) {"name": "metadata", "type": "tag"}, # Using TAG instead of TEXT for Valkey compatibility {"name": "created_at", "type": "numeric"}, {"name": "updated_at", "type": "numeric"}, @@ -169,7 +169,7 @@ class ValkeyDB(VectorStoreBase): "user_id", "TAG", "memory", - "TAG", + "TEXT", "metadata", "TAG", "created_at", diff --git a/tests/vector_stores/test_valkey.py b/tests/vector_stores/test_valkey.py index 78d66615e..715b1bb2a 100644 --- a/tests/vector_stores/test_valkey.py +++ b/tests/vector_stores/test_valkey.py @@ -1020,3 +1020,44 @@ def test_cluster_mode_delete(valkey_db_cluster, mock_valkey_cluster_client): """Test that delete works in cluster mode.""" valkey_db_cluster.delete("id1") mock_valkey_cluster_client.delete.assert_called_once_with("mem0:test_cluster:id1") + + +# Schema field-type tests (regression for #5006) + + +def test_default_fields_memory_is_text(): + """The 'memory' field stores free-form memory text, so it must be indexed as + TEXT (full-text searchable, tokenized) rather than TAG. + + Regression test for #5006: TAG only tokenizes on commas, so the entire memory + text collapses into a single token, breaking full-text search and bloating the + inverted index. + """ + from mem0.vector_stores.valkey import DEFAULT_FIELDS + + memory_field = next(f for f in DEFAULT_FIELDS if f["name"] == "memory") + assert memory_field["type"] == "text", ( + "DEFAULT_FIELDS['memory'] must be 'text' (full-text searchable), not 'tag'" + ) + + +def test_build_index_schema_indexes_memory_as_text(valkey_db): + """The FT.CREATE command emitted by _build_index_schema must declare the + 'memory' field as TEXT, not TAG. + + Regression test for #5006. + """ + cmd = valkey_db._build_index_schema( + collection_name="test_collection", + embedding_dims=1536, + distance_metric="COSINE", + prefix="mem0:test_collection", + ) + + # Locate the 'memory' field declaration and check the type token that follows it. + memory_idx = cmd.index("memory") + assert cmd[memory_idx + 1] == "TEXT", ( + f"'memory' field must be indexed as TEXT, got {cmd[memory_idx + 1]!r}" + ) + # And it must not be declared as TAG. + assert ["memory", "TAG"] != cmd[memory_idx : memory_idx + 2]