From 6ed5fcf76e422e40a24bf1e6f1d64aad41570676 Mon Sep 17 00:00:00 2001 From: kartik-mem0 Date: Thu, 21 May 2026 17:43:01 +0530 Subject: [PATCH] feat(mem0-plugin): expand categories to 17, add recall rubric + confidence + files tagging Tier 5 implementation: - #17: Expand coding categories from 7 to 17 (dependency_decisions, performance_findings, security_constraints, testing_patterns, data_model, api_contracts, deployment_runbook, team_norms, domain_glossary, experiment_results) - #18: Category-recall rubric table in mem0-mcp/SKILL.md mapping user intent to 2-3 categories for parallel search fan-out - #19: metadata.confidence scoring rules (1.0/0.8/0.5/0.3 scale) - #20: metadata.files array tagging for affected file paths --- .../scripts/setup_coding_categories.py | 60 +++++++++++++ mem0-plugin/skills/mem0-mcp/SKILL.md | 70 +++++++++++++++ mem0-plugin/tests/test_coding_categories.py | 88 +++++++++++++++++++ 3 files changed, 218 insertions(+) create mode 100644 mem0-plugin/tests/test_coding_categories.py diff --git a/mem0-plugin/scripts/setup_coding_categories.py b/mem0-plugin/scripts/setup_coding_categories.py index 81a5d1cec..b3cd936b4 100644 --- a/mem0-plugin/scripts/setup_coding_categories.py +++ b/mem0-plugin/scripts/setup_coding_categories.py @@ -69,6 +69,66 @@ CODING_CATEGORIES = [ "and ways of working." ) }, + { + "dependency_decisions": ( + "Why specific libraries, frameworks, or package versions were chosen or replaced, " + "including the alternatives considered and the reasoning behind the selection." + ) + }, + { + "performance_findings": ( + "Profiling results, bottlenecks identified, optimisations applied, and measurable " + "improvements achieved -- useful for avoiding regressions and guiding future work." + ) + }, + { + "security_constraints": ( + "Security requirements, authentication and authorisation rules, data-handling " + "constraints, compliance obligations, and known threat mitigations in effect." + ) + }, + { + "testing_patterns": ( + "Test strategies, frameworks chosen, coverage targets, fixture patterns, mocking " + "approaches, and how the test suite is structured for this project." + ) + }, + { + "data_model": ( + "Schema definitions, database column semantics, domain object relationships, " + "field constraints, and how data flows between storage and application layers." + ) + }, + { + "api_contracts": ( + "API endpoint shapes, request and response schemas, authentication requirements, " + "versioning policy, and any breaking-change commitments or deprecation timelines." + ) + }, + { + "deployment_runbook": ( + "How to build, release, deploy, and roll back the project. CI/CD pipeline steps, " + "environment-specific configuration, and on-call runbook entries." + ) + }, + { + "team_norms": ( + "Team working agreements, PR review etiquette, branching strategy, on-call " + "rotation, and other social or process conventions the team has agreed on." + ) + }, + { + "domain_glossary": ( + "Domain-specific terms, abbreviations, and acronyms with their precise meanings " + "in this project -- prevents misunderstandings across code, docs, and discussion." + ) + }, + { + "experiment_results": ( + "Results from A/B tests, feature-flag experiments, spikes, or proof-of-concept " + "work -- what was tried, what was measured, and what conclusion was reached." + ) + }, ] diff --git a/mem0-plugin/skills/mem0-mcp/SKILL.md b/mem0-plugin/skills/mem0-mcp/SKILL.md index 9ae3fc3db..a5b2e334d 100644 --- a/mem0-plugin/skills/mem0-mcp/SKILL.md +++ b/mem0-plugin/skills/mem0-mcp/SKILL.md @@ -75,6 +75,30 @@ Combine `user_id` + `app_id` with one metadata clause per call: | `{"metadata": {"type": "user_preference"}}` | tooling, stack, style — always include for code work | | `{"metadata": {"type": "convention"}}` | established patterns in this project | +### Which categories to search by query intent + +When a query clearly maps to one of the platform's custom categories, fan-out to 2–3 parallel `search_memories` calls scoped to those categories so recall is precise without being noisy. Use the `metadata.type` filter as your primary discriminator; treat the category column below as the semantic lens to pick the right query nouns. + +| User intent / signal | Primary categories to search | Example query nouns | +|---|---|---| +| Design or architecture question | `architecture_decisions`, `api_contracts`, `data_model` | `"architecture decision"`, `"API schema"`, `"data model"` | +| Something failed / debugging | `anti_patterns`, `bug_fixes`, `security_constraints` | `"bug root cause"`, `"failure pattern"`, `"security constraint"` | +| How do we do X here? | `coding_conventions`, `team_norms`, `testing_patterns` | `"code convention"`, `"team norm"`, `"test strategy"` | +| Which library / version to use | `dependency_decisions`, `tooling_setup`, `architecture_decisions` | `"dependency choice"`, `"library version"`, `"tooling setup"` | +| Performance or scale concern | `performance_findings`, `architecture_decisions`, `data_model` | `"performance bottleneck"`, `"profiling result"`, `"optimisation"` | +| Security / auth / compliance | `security_constraints`, `api_contracts`, `coding_conventions` | `"auth rule"`, `"security requirement"`, `"compliance"` | +| Test strategy or coverage | `testing_patterns`, `coding_conventions`, `anti_patterns` | `"test framework"`, `"coverage target"`, `"fixture pattern"` | +| Schema / DB / domain object | `data_model`, `api_contracts`, `domain_glossary` | `"schema"`, `"column"`, `"domain object"` | +| API shape or versioning | `api_contracts`, `data_model`, `architecture_decisions` | `"endpoint"`, `"request schema"`, `"versioning"` | +| How to deploy / release / rollback | `deployment_runbook`, `tooling_setup`, `team_norms` | `"deploy step"`, `"rollback"`, `"CI pipeline"` | +| Team process / branching / PRs | `team_norms`, `coding_conventions`, `deployment_runbook` | `"branching strategy"`, `"PR review"`, `"working agreement"` | +| What does this term mean? | `domain_glossary`, `data_model`, `api_contracts` | `"glossary"`, `"abbreviation"`, `"domain term"` | +| Experiment / spike / A-B test | `experiment_results`, `performance_findings`, `anti_patterns` | `"experiment result"`, `"A/B test"`, `"spike outcome"` | +| User's tool / language preferences | `user_preferences`, `tooling_setup`, `coding_conventions` | `"user preference"`, `"preferred tool"`, `"language choice"` | +| Past task strategies that worked | `task_learnings`, `anti_patterns`, `coding_conventions` | `"task strategy"`, `"approach that worked"` | +| Environment / setup question | `tooling_setup`, `deployment_runbook`, `dependency_decisions` | `"environment setup"`, `"build tool"`, `"install step"` | +| Anything related to current state | `task_learnings`, `architecture_decisions`, `anti_patterns` | (combine with recency filter — see below) | + Full filter (replace `` and `` with the active values from SessionStart): ```python filters={"AND": [{"user_id": ""}, {"app_id": ""}, {"metadata": {"type": "decision"}}]} @@ -187,3 +211,49 @@ Include metadata: `{"type": "session_state"}` - Do NOT write to MEMORY.md or any file-based memory. Use mem0 MCP tools exclusively. - Only store genuinely useful learnings. Skip trivial interactions. - Use specific, searchable language in memory content. + +### Confidence scoring on every add_memory + +Every `add_memory` call MUST include a `confidence` field in its `metadata` object. This captures how certain the stored fact is, so downstream callers can filter out speculation. + +| `metadata.confidence` value | Meaning | When to use | +|---|---|---| +| `1.0` | User explicitly stated it | User said "we use Postgres", "always lint before commit", "never use floats for currency" | +| `0.8` | Observed directly in code / config | You read it from a file, migration, or config — not inferred | +| `0.5` | Inferred from context | You derived it from surrounding evidence but the user didn't confirm it | +| `0.3` | Guessed / low-signal | Extrapolated from a single weak signal; treat as a tentative hypothesis | + +Example: + +```python +add_memory( + messages=[{"role": "user", "content": "We always use Postgres — never SQLite in production."}], + user_id="", + app_id="", + metadata={"type": "architecture_decisions", "branch": "", "confidence": 1.0}, + infer=False, +) +``` + +**Search guidance:** When recalling actionable facts (decisions, conventions, security constraints), optionally apply a confidence threshold of 0.6 or above to avoid surfacing low-confidence guesses. Only top-level metadata keys are filterable, so `confidence` filtering requires SDK-side post-filtering or a dedicated high-confidence write path — for now, include the confidence value in every write and document it in the memory content so it is searchable via text. + +### File path tagging on every add_memory + +Every `add_memory` call that is associated with specific files MUST include a `files` key in its `metadata` object. The value is an array of affected file paths relative to the project root. + +```python +add_memory( + messages=[{"role": "user", "content": "The auth middleware lives in src/middleware/auth.ts and validates JWTs using the shared key in config/secrets.ts."}], + user_id="", + app_id="", + metadata={ + "type": "architecture_decisions", + "branch": "", + "confidence": 0.8, + "files": ["src/middleware/auth.ts", "config/secrets.ts"], + }, + infer=False, +) +``` + +**Filtering note:** The mem0 v2 filter API does not yet support `array-contains` predicates. You cannot filter by `metadata.files` at search time. To work around this, always embed the bare filenames (and important path segments) in the memory content text itself — the vector search will then surface them on a filename query. The `files` array in metadata is still written for future compatibility once array-contains filtering is available. diff --git a/mem0-plugin/tests/test_coding_categories.py b/mem0-plugin/tests/test_coding_categories.py new file mode 100644 index 000000000..c4d57b62f --- /dev/null +++ b/mem0-plugin/tests/test_coding_categories.py @@ -0,0 +1,88 @@ +"""Tests for setup_coding_categories.py -- CODING_CATEGORIES list completeness.""" + +from __future__ import annotations + +import importlib +import os +import sys + +import pytest + +SCRIPTS_DIR = os.path.join(os.path.dirname(__file__), "..", "scripts") + +EXPECTED_KEYS = [ + "architecture_decisions", + "anti_patterns", + "task_learnings", + "tooling_setup", + "bug_fixes", + "coding_conventions", + "user_preferences", + "dependency_decisions", + "performance_findings", + "security_constraints", + "testing_patterns", + "data_model", + "api_contracts", + "deployment_runbook", + "team_norms", + "domain_glossary", + "experiment_results", +] + + +@pytest.fixture() +def coding_categories(): + """Import CODING_CATEGORIES from setup_coding_categories, ensuring scripts/ is on path.""" + abs_scripts = os.path.abspath(SCRIPTS_DIR) + inserted = False + if abs_scripts not in sys.path: + sys.path.insert(0, abs_scripts) + inserted = True + # Force re-import in case another test already loaded a stale version + mod_name = "setup_coding_categories" + if mod_name in sys.modules: + del sys.modules[mod_name] + mod = importlib.import_module(mod_name) + yield mod.CODING_CATEGORIES + if inserted and abs_scripts in sys.path: + sys.path.remove(abs_scripts) + + +def test_total_count(coding_categories): + """CODING_CATEGORIES must contain exactly 17 entries.""" + assert len(coding_categories) == 17, ( + f"Expected 17 categories, found {len(coding_categories)}: " + f"{[list(c.keys())[0] for c in coding_categories]}" + ) + + +def test_all_expected_keys_present(coding_categories): + """Every expected category key must appear exactly once.""" + actual_keys = [list(cat.keys())[0] for cat in coding_categories] + for key in EXPECTED_KEYS: + assert key in actual_keys, f"Missing expected category key: '{key}'" + + +def test_no_duplicate_keys(coding_categories): + """No category key may appear more than once.""" + actual_keys = [list(cat.keys())[0] for cat in coding_categories] + seen = set() + duplicates = [] + for key in actual_keys: + if key in seen: + duplicates.append(key) + seen.add(key) + assert not duplicates, f"Duplicate category keys found: {duplicates}" + + +def test_each_description_is_non_empty_string(coding_categories): + """Every category must have a non-empty string description.""" + for cat in coding_categories: + assert len(cat) == 1, f"Category dict should have exactly one key, got: {cat}" + key = list(cat.keys())[0] + description = cat[key] + assert isinstance(description, str), ( + f"Category '{key}' description is not a string: {type(description)}" + ) + assert description.strip(), f"Category '{key}' has an empty description"