diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 8b0e16623..3bf63cfbd 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -12,7 +12,7 @@ "name": "mem0", "source": "./mem0-plugin", "description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search to Claude workflows.", - "version": "0.1.1" + "version": "0.1.2" } ] } diff --git a/AGENTS.md b/AGENTS.md index 1fe035342..52d8ffc1e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -27,7 +27,7 @@ This is a **polyglot monorepo** containing Python and TypeScript packages, CLIs, | `server/` | FastAPI REST server for self-hosted Mem0 (Docker: FastAPI + PostgreSQL/pgvector + Neo4j) | | `openmemory/` | Self-hosted memory platform — `api/` (FastAPI + Alembic + MCP server) and `ui/` (Next.js 15 + React 19) | | `mem0-plugin/` | AI editor plugins (Claude Code, Cursor, Codex) — MCP server connection, lifecycle hooks, skills | -| `skills/` | Claude Code skill definitions — `mem0/`, `mem0-cli/`, `mem0-vercel-ai-sdk/` | +| `skills/` | Claude Code skill definitions. Reference skills (SDK knowledge, always-on): `mem0/`, `mem0-cli/`, `mem0-vercel-ai-sdk/`. Pipeline skills (run on demand): `mem0-integrate/`, `mem0-test-integration/` | | `docs/` | Documentation site (Mintlify) | | `tests/` | Python SDK tests (pytest) | | `evaluation/` | Benchmarking framework — LOCOMO evals, experiment runner, score generation | @@ -387,7 +387,9 @@ Model Context Protocol support in multiple places: ### Plugin & Skills System - `mem0-plugin/` provides integrations for Claude Code, Cursor, and Codex via MCP server connections and lifecycle hooks for automatic memory capture. -- `skills/` contains structured skill definitions for AI agents, covering SDK usage, CLI workflows, and Vercel AI SDK patterns. +- `skills/` contains structured skill definitions for AI agents, split into two categories: + - **Reference skills** (always-on SDK knowledge): `mem0` (Python + TS SDKs, framework integrations), `mem0-cli` (terminal workflows), `mem0-vercel-ai-sdk` (Vercel AI provider). + - **Pipeline skills** (run on demand): `mem0-integrate` wires Mem0 into an existing repo via a TDD pipeline; `mem0-test-integration` verifies what the integrator produced on the same branch. The two are loosely coupled via `.mem0-integration/` artifacts. ### Adding a New Provider diff --git a/README.md b/README.md index 9eef64f7c..1dea670c0 100644 --- a/README.md +++ b/README.md @@ -147,6 +147,27 @@ mem0 search "What does Alice prefer?" --user-id alice See the [CLI documentation](https://docs.mem0.ai/platform/cli) for the full command reference. +### Agent Skills + +Teach your AI coding assistant (Claude Code, Codex, Cursor, Windsurf, OpenCode, OpenClaw, and any tool that supports the skills standard) how to build with Mem0. Two categories: + +**Reference skills — always on** (SDK knowledge loaded into the assistant's context): + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0 +npx skills add https://github.com/mem0ai/mem0 --skill mem0-cli +npx skills add https://github.com/mem0ai/mem0 --skill mem0-vercel-ai-sdk +``` + +**Pipeline skills — run on demand** (execute an end-to-end workflow in an existing repo): + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate +npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration +``` + +Use `/mem0-integrate` to wire Mem0 into an existing repo via a test-first pipeline, then `/mem0-test-integration` to verify. See the [skills catalog](./skills/) or [Vibecoding with Mem0](https://docs.mem0.ai/vibecoding) for the full picture. + ### Basic Usage Mem0 requires an LLM to function, with `gpt-5-mini` from OpenAI as the default. However, it supports a variety of LLMs; for details, refer to our [Supported LLMs documentation](https://docs.mem0.ai/components/llms/overview). diff --git a/docs/api-reference/organizations-projects.mdx b/docs/api-reference/organizations-projects.mdx index cbdf1f162..d4958b260 100644 --- a/docs/api-reference/organizations-projects.mdx +++ b/docs/api-reference/organizations-projects.mdx @@ -109,6 +109,19 @@ client.project.update( ) ``` +#### Toggle Memory Decay + +`decay` is a per-project boolean that turns on [Memory Decay](/platform/features/memory-decay) — a search-time ranking bias that reinforces recently-accessed memories and gently dampens stale ones. The flag is `false` by default; set it via the same project-update endpoint: + +```bash cURL +curl -X PATCH https://api.mem0.ai/api/v1/orgs/organizations/$ORG_ID/projects/$PROJECT_ID/ \ + -H "Authorization: Token $MEM0_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"decay": true}' +``` + +The current state is returned on every project read (and supports `?fields=decay` for a minimal response). Toggling has no effect on stored memories, only on how v3 search ranks them. + ### Delete Project diff --git a/docs/changelog/highlights.mdx b/docs/changelog/highlights.mdx index 296017ed0..7739e1c09 100644 --- a/docs/changelog/highlights.mdx +++ b/docs/changelog/highlights.mdx @@ -10,9 +10,8 @@ mode: "wide" Mem0 Platform v3 can now interpret time-aware memories and queries so assistants retrieve the right information for questions about the past, upcoming plans, and current state. -- **Time-aware search intent** — Queries like `last week`, `upcoming`, `right now`, and `as of March 2025` use temporal signals during retrieval -- **Write-time temporal enrichment** — Dated events, future plans, ongoing states, relationships, and preferences are handled automatically when temporal structure is detectable -- **Enabled by default** — No per-request toggle is required for v3 writes or searches +- **Time-aware search intent** — Queries like `last week`, `upcoming`, `right now`, and `as of March 2025` return contextually appropriate results automatically +- **Enabled by default** — No per-request toggle required for v3 writes or searches - **Anchored relative queries** — `reference_date` anchors relative search phrases for tests, backfills, and reproducible demos - **Normal response shape** — Temporal reasoning affects ranking while preserving existing client response patterns @@ -20,6 +19,19 @@ See [Temporal Reasoning](/platform/features/temporal-reasoning) for usage detail + + +**Memory Decay — Recently-Used Memories Surface Higher, Automatically** + +Per-project search-time ranking bias that boosts recently-touched memories and gently dampens stale ones. Off by default; opt in per project via the `decay` field on the project endpoint, or via `client.project.update(decay=True)` in the SDKs (Python `v2.0.2` / TypeScript `v3.0.3`). + +- **Soft bias, never a filter.** The scaling factor stays in `0.3×–1.5×`. Decay can reorder candidates but never zeros them out — anything that surfaced before decay can still surface after. +- **Reinforcement loop.** Every memory returned in a search has its access history updated, so frequently-used facts naturally float to the top over time. +- **Public score still clamped to `[0, 1]`.** Existing API contract preserved; no client-side changes needed. +- **v3 search only**, fully reversible. See [Memory Decay docs](/platform/features/memory-decay). + + + **New Memory Algorithm — State-of-the-Art Accuracy at ~3-4x Lower Cost** diff --git a/docs/changelog/platform.mdx b/docs/changelog/platform.mdx index b3c59ebd6..6547f3124 100644 --- a/docs/changelog/platform.mdx +++ b/docs/changelog/platform.mdx @@ -11,11 +11,17 @@ mode: "wide" - **Search:** Added `reference_date` support to anchor relative temporal queries for tests, backfills, and reproducible demos **Improvements:** -- **Memory:** Temporal enrichment runs asynchronously by default after v3 writes and feeds retrieval ranking once available - **API:** Temporal reasoning preserves the normal client response shape for search and get-all results + + +**New Features:** +- **Memory Decay:** Per-project search-time ranking bias that boosts recently-used memories and gently dampens stale ones. Opt-in via `decay` on the project endpoint; off by default. The scaling factor stays in `0.3×–1.5×`, the public `score` remains clamped to `[0, 1]`, and the bias never filters a candidate out. See [Memory Decay docs](/platform/features/memory-decay). + + + **Improvements:** diff --git a/docs/changelog/sdk.mdx b/docs/changelog/sdk.mdx index ede4cf117..c9db329fe 100644 --- a/docs/changelog/sdk.mdx +++ b/docs/changelog/sdk.mdx @@ -7,6 +7,20 @@ mode: "wide" + + +**Bug Fixes:** +- **Telemetry:** Stitch OSS and platform PostHog identities on `MemoryClient` init so `$identify` events fire and a single user is no longer tracked as two or three disconnected personas ([#5040](https://github.com/mem0ai/mem0/pull/5040)) +- **Security:** Harden against SQL injection and prompt injection ([#4997](https://github.com/mem0ai/mem0/pull/4997)) + +**New Features:** +- **SDK:** Expose `decay` on `project.update` ([#5062](https://github.com/mem0ai/mem0/pull/5062)) + +**Improvements:** +- **Plugin:** Hand `mem0` search decisions to the agent ([#4992](https://github.com/mem0ai/mem0/pull/4992)) + + + **Bug Fixes:** @@ -910,6 +924,18 @@ See the [OSS v1 to v2 migration guide](https://docs.mem0.ai/migration/oss-v1-to- + + +**Bug Fixes:** +- **Telemetry:** Stitch OSS and platform PostHog identities on `MemoryClient` init so `$identify` events fire and a single user is no longer tracked as two or three disconnected personas ([#5040](https://github.com/mem0ai/mem0/pull/5040)) +- **Vector Stores:** Fix inverted vector distance in PGVector implementation ([#4944](https://github.com/mem0ai/mem0/pull/4944)) +- **Security:** Harden against SQL injection and prompt injection ([#4997](https://github.com/mem0ai/mem0/pull/4997)) + +**New Features:** +- **SDK:** Expose `decay` on `project.update` ([#5062](https://github.com/mem0ai/mem0/pull/5062)) + + + **Bug Fixes:** diff --git a/docs/docs.json b/docs/docs.json index 4bbf062cb..e3644a7ae 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -84,7 +84,8 @@ "platform/advanced-memory-operations", "platform/features/criteria-retrieval", "platform/features/contextual-add", - "platform/features/custom-instructions" + "platform/features/custom-instructions", + "platform/features/memory-decay" ] }, { diff --git a/docs/llms.txt b/docs/llms.txt index 2fe77074c..7ef58b201 100644 --- a/docs/llms.txt +++ b/docs/llms.txt @@ -188,6 +188,7 @@ If the user is on a pre-current major (Python < 2, TS < 3, or Platform `output_f - [Temporal Reasoning](https://docs.mem0.ai/platform/features/temporal-reasoning) [Platform]: Use when time-aware searches like last week, upcoming, or right now need better result ordering. - [Contextual Add](https://docs.mem0.ai/platform/features/contextual-add) [Platform]: Use when `add()` should consider the surrounding conversation, not just the latest turn. - [Custom Instructions](https://docs.mem0.ai/platform/features/custom-instructions) [Platform]: Use when tailoring what Mem0 extracts and stores on Platform. +- [Memory Decay](https://docs.mem0.ai/platform/features/memory-decay) [Platform]: Use when search results should boost recently-reinforced memories and dampen stale ones — opt-in per project, search-time only, never filters candidates out. - [Advanced Memory Operations](https://docs.mem0.ai/platform/advanced-memory-operations) [Platform]: Use when basic CRUD is not enough - batch ops, complex filters, workflows. ### Features - Data Management diff --git a/docs/platform/features/memory-decay.mdx b/docs/platform/features/memory-decay.mdx new file mode 100644 index 000000000..4c4cfd37d --- /dev/null +++ b/docs/platform/features/memory-decay.mdx @@ -0,0 +1,191 @@ +--- +title: Memory Decay +description: "Boost recently-used memories and gently dampen stale ones at search time, without filtering anything out." +--- + +# Memory Decay + +Older memories drift in relevance at different speeds. A user's coffee order matters every morning; a one-off project name from last quarter rarely matters again. Memory Decay makes that intuition explicit at search time: every time a memory is returned in a search it gets a small reinforcement, and memories that haven't been touched in a while have their ranking score gently dampened. + +It is **a soft ranking bias, never a filter.** Decay never zeroes a candidate out — at worst it scales its score by `0.3×`. Anything that would have surfaced without decay can still surface with decay on, just with a different ranking among similarly-scored results. + + + **Use Memory Decay when…** + - Search results are crowded with old facts the user no longer cares about. + - You want recently-used memories to drift to the top automatically — without writing custom scoring logic. + - You want this preference applied per project so cohorts can be compared side-by-side. + + + + Memory Decay is **opt-in per project** and **off by default**. Search behavior is bit-identical to today until you turn it on. The toggle applies to v3 search only. + + +## How it works + +Every memory carries a small piece of bookkeeping: when was it last retrieved, and how often. Memory Decay turns that history into a *scaling factor* in the range `0.3×` to `1.5×` and multiplies it into the ranking score at search time. + +| Memory state | Scaling factor | Ranking effect | +|---|---|---| +| Just accessed | ≈ **1.5×** | Strong boost | +| Touched today | 1.2 – 1.4× | Mild boost | +| Idle for a few days | 0.6 – 1.0× | Mild dampening | +| Idle for weeks | 0.4 – 0.6× | Stronger dampening | +| Idle for many months / years | ≈ **0.3×** | Floor — never lower | + +The bounds matter: `0.3` is the floor and `1.5` is the ceiling, so decay can meaningfully reorder candidates without ever dominating the underlying relevance score. + +At search time the pipeline: + +1. Widens the candidate pool (`top_k × 3`, with a floor of 50) so reordering has room. +2. Multiplies each candidate's score by its scaling factor. +3. Sorts on the unclamped product so the full `0.3×–1.5×` range can rearrange candidates. +4. Returns the public `score` clamped to `[0, 1]` so the API contract is preserved. +5. Truncates to the `top_k` you requested. +6. Records a fire-and-forget reinforcement against each returned memory — its access history grows by one, capped at the most recent 20 touches. + +Memories created before decay was enabled don't yet have an access history. They use a sensible fallback: their `updated_at` is treated as a single past touch, so the same scale above applies based on how stale that update is — a recently-updated legacy memory enters near the neutral band, a long-stale one sits closer to the floor. Once surfaced in a search after decay is on, they accumulate access history naturally and behave like any other memory. + +## Configure access + +- Set `MEM0_API_KEY` in your environment, or pass it to the SDK constructor. +- Initialize the client with the organization and project you want to scope to. + +The toggle lives on the project. You enable decay by patching the project's `decay` field; everything else — your `add` calls, your `search` calls, your application code — stays exactly the same. + +## Enable decay for a project + +### 1. Turn the flag on + +The toggle is exposed on the standard project-update endpoint, the same place where `multilingual` and `custom_categories` live. + + +```python Python +client.project.update(decay=True) +``` + +```javascript JavaScript +await client.project.update({ decay: true }); +``` + +```bash cURL +curl -X PATCH https://api.mem0.ai/api/v1/orgs/organizations/$ORG_ID/projects/$PROJECT_ID/ \ + -H "Authorization: Token $MEM0_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"decay": true}' +``` + +```json Response +{ "message": "Updated decay" } +``` + + +### 2. Confirm the state + +`decay` is returned on every project read. To fetch only this field, use `?fields=decay`. + + +```python Python +response = client.project.get(fields=["decay"]) +print(response["decay"]) +``` + +```javascript JavaScript +const response = await client.project.get({ fields: ["decay"] }); +console.log(response.decay); +``` + +```bash cURL +curl "https://api.mem0.ai/api/v1/orgs/organizations/$ORG_ID/projects/$PROJECT_ID/?fields=decay" \ + -H "Authorization: Token $MEM0_API_KEY" +``` + +```json Response +{ "decay": true } +``` + + +### 3. Turn it back off + +The toggle is fully reversible. Setting it to `false` immediately restores the pre-decay ranking; nothing about your stored memories is modified or lost. + + +```python Python +client.project.update(decay=False) +``` + +```javascript JavaScript +await client.project.update({ decay: false }); +``` + +```bash cURL +curl -X PATCH https://api.mem0.ai/api/v1/orgs/organizations/$ORG_ID/projects/$PROJECT_ID/ \ + -H "Authorization: Token $MEM0_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"decay": false}' +``` + + + + The toggle is idempotent. Re-applying the same value is a no-op, and access history accumulated while decay was on is preserved if you flip it back on later. + + +## What changes when decay is on + +- **Search ranking reorders.** A relevant memory you reinforced an hour ago will tend to outrank an equally-relevant memory that was last touched a month ago. +- **The candidate pool over-fetches** to give the scaling factor room to reorder. You still get exactly the `top_k` you requested, but the items returned can come from a deeper slice of the pre-decay ranking than before. +- **The public `score` field stays in `[0, 1]`.** Even when the internal product exceeds 1, the field returned to the client is clamped, so existing assertions and downstream UI logic continue to work. + +## What stays the same + +- **Public API shape** — every endpoint accepts the same parameters and returns the same fields. You don't touch your client code. +- **Threshold semantics on the request side** — your `threshold` is still applied during candidate selection. +- **Memory creation and storage** — every new memory still lands the same way. Decay is a search-time concern. +- **Per-memory data** — categories, metadata, timestamps, embeddings: untouched. + + + Because the scaling factor is applied *after* the threshold filter has already run, an item that passed the request `threshold` can come back with a public `score` slightly below it (a stale candidate dampened by `0.3×`). This is intentional — decay is a soft bias, not a filter. If you require a hard `score >= threshold` invariant on the response, filter client-side after the call. + + +## Lifecycle of a memory under decay + +| Stage | Scaling factor | Effect | +|---|---|---| +| Just added | ≈ 1.5× | Strong boost — fresh facts surface easily. | +| Reinforced on a recent search | 1.2 – 1.5× | Sustains its boost for the next several searches. | +| Idle for a few days | 0.6 – 1.0× | Falls back into the neutral band. | +| Idle for weeks | 0.4 – 0.6× | Mild dampening — can still surface for strong matches. | +| Pre-decay legacy memory (no access history) | 0.3 – 1.0× | Falls back to `updated_at`: recently-updated entries land near 1.0×, long-stale entries approach the 0.3× floor. | + +The reinforcement is bounded: each memory tracks at most the last 20 access timestamps, so the boost stays well-behaved no matter how many times a memory is retrieved. + +## FAQ + +**Will decay ever drop a result that would otherwise surface?** +No. The floor is `0.3×` — the scaling factor can dampen a score, never zero it. Threshold filtering happens *before* decay, so any candidate that cleared the threshold is in the pool decay reorders. + +**Why is the public score sometimes below my requested threshold?** +The threshold is applied to the candidate pool pre-decay; the scaling factor then reshapes scores in the `0.3×–1.5×` band. A stale-but-relevant candidate can come back with a final score slightly under your threshold by design — the candidate stays visible but visibly dampened. Filter client-side if you need a hard floor on the response. + +**Does decay change how I add memories?** +No. The `client.add(...)` path is unchanged. Decay is a search-time ranking adjustment. + +**What if I had memories before turning decay on?** +They use a fallback: the memory's `updated_at` is treated as a single historical touch, so the same scaling applies based on how stale that update is — a recently-updated legacy memory enters near the neutral band (~1.0×), a long-stale one closer to the floor (~0.3×). Once retrieved they accumulate access history and behave like any other memory. + +**Can I tune how aggressively decay scales scores?** +Not in this version. The current scaling is calibrated to be conservative — wide enough to meaningfully reorder candidates, narrow enough to never dominate the underlying relevance score. Per-project tuning is on the roadmap. + +**Can I see the scaling factor per result?** +Internal scoring details are persisted on the search Event for support and debugging. They aren't exposed in the public response by design — the response surface stays a single `score` field. + +**Does decay interact with reranking?** +Yes — they layer cleanly. The reranker produces a richer relevance score; decay then biases that score by reinforcement history before final truncation to `top_k`. + +## What's next + +This release is deliberately the simplest version of decay we could ship — every memory contributes to ranking through its access history alone, so the signal can be evaluated in isolation. On the roadmap: + +- **Category-aware weighting.** A fact tagged `health` will be able to carry more weight than a passing observation tagged `misc`, so important categories don't get dampened the same way as noise. +- **Auto-tuning per project.** Project-scoped automatic adjustment of how aggressively decay scales scores, based on observed access patterns — replacing the fixed scaling band with one that fits your workload. + +Both extensions are forward-compatible — no migration on your side will be needed when they ship. diff --git a/docs/vibecoding.mdx b/docs/vibecoding.mdx index 07949f437..118f587ec 100644 --- a/docs/vibecoding.mdx +++ b/docs/vibecoding.mdx @@ -22,13 +22,35 @@ We follow the llms.txt standard: ## Agent Skills -Teach your coding assistant how to build with Mem0: +Mem0 ships two kinds of skills for AI coding assistants. Both work with Claude Code, Codex, Cursor, Windsurf, OpenCode, OpenClaw, and any assistant that supports the skills standard. + +### Reference skills — always on + +Teach your assistant Mem0's SDK surface so it writes correct code in everyday development: ```bash npx skills add https://github.com/mem0ai/mem0 --skill mem0 +npx skills add https://github.com/mem0ai/mem0 --skill mem0-cli +npx skills add https://github.com/mem0ai/mem0 --skill mem0-vercel-ai-sdk ``` -Works with Claude Code, Cursor, Windsurf, and any assistant that supports skills. Once installed, your assistant understands Mem0's full API, framework integrations, and common patterns. +- `mem0` — Python and TypeScript SDKs (Platform + OSS), plus framework integrations (LangChain, CrewAI, OpenAI Agents, LangGraph, LlamaIndex, etc.) +- `mem0-cli` — terminal workflows for the `mem0` CLI (both Node and Python builds) +- `mem0-vercel-ai-sdk` — `@mem0/vercel-ai-provider` and `createMem0` + +### Pipeline skills — run on demand + +Let your assistant execute an end-to-end workflow in an existing repo. Invoked as slash commands: + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate +npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration +``` + +- `/mem0-integrate` — wire Mem0 into an existing repository using a goal-driven, test-first pipeline. Detects the stack, asks whether to use Platform or OSS, writes failing tests first, and keeps the integration additive and feature-flagged. +- `/mem0-test-integration` — verify what `/mem0-integrate` produced. Runs the repo's native test suite and a real end-to-end smoke flow against your API key, then produces a scorecard. + +See the [skills index](https://github.com/mem0ai/mem0/tree/main/skills) for the full catalog. ## MCP Server Setup diff --git a/mem0-plugin/.claude-plugin/plugin.json b/mem0-plugin/.claude-plugin/plugin.json index 55820ea6f..0eb499f98 100644 --- a/mem0-plugin/.claude-plugin/plugin.json +++ b/mem0-plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "mem0", - "version": "0.1.1", + "version": "0.1.2", "description": "Mem0 memory layer for AI applications. Add persistent memory, personalization, and semantic search to Claude workflows using the Mem0 Platform MCP server.", "author": { "name": "Mem0", diff --git a/mem0-plugin/README.md b/mem0-plugin/README.md index 12845318f..732c44681 100644 --- a/mem0-plugin/README.md +++ b/mem0-plugin/README.md @@ -157,6 +157,32 @@ After installing, confirm the MCP server is connected: - **Mem0 SDK Skill** — Guides the AI on how to integrate the Mem0 SDK (Python & TypeScript) into your applications. - **Memory Protocol Skill** — Codex-specific skill that instructs the agent to retrieve relevant memories at task start, store learnings on completion, and capture session state before context loss. Complements the lifecycle hooks on Codex. +## Updating the plugin + +When the plugin updates (new version pulled from the marketplace, or a fresh local install), the MCP server connection in your existing Claude Code / Cursor / Codex session is left holding a stale handle and stops responding. **Restart your client to reconnect:** + +- **Claude Code:** run `/restart` in the prompt, or close and reopen the CLI. +- **Cursor:** quit and relaunch. +- **Codex:** restart the editor session. + +Your `MEM0_API_KEY` doesn't need to be re-entered — the auth header is re-read from your environment on the new session. The plugin's MCP config uses `${MEM0_API_KEY}` interpolation at session start, not at install time, so as long as the env var is set persistently (in your shell profile or `~/.claude/settings.json` `env` block), reconnection is automatic on restart. + +If reconnection still fails after a restart, check that `MEM0_API_KEY` is reachable in the new shell (`echo $MEM0_API_KEY`) and confirm you're using a key that starts with `m0-` (from https://app.mem0.ai/dashboard/api-keys, not a legacy token). + +## Optional: tune categories for coding workflows + +mem0 auto-tags every memory with one or more `categories` from a project-level list. The default list is consumer-oriented (`food`, `hobbies`, `music` …) — useful for chat assistants, less so for code. A one-shot script in this plugin replaces it with a coding-focused taxonomy: + +```bash +# Dry-run first -- prints current vs proposed, no changes: +python mem0-plugin/scripts/setup_coding_categories.py + +# Actually write: +python mem0-plugin/scripts/setup_coding_categories.py --apply +``` + +Requires the `mem0ai` Python SDK (`pip install mem0ai`) and `MEM0_API_KEY` set. New memories will then auto-tag against `architecture_decisions`, `anti_patterns`, `task_learnings`, `tooling_setup`, `bug_fixes`, `coding_conventions`, `user_preferences`. Re-run with a different list any time; `project.update(custom_categories=[...])` always replaces. + ## MCP Tools Once installed, the following tools are available: diff --git a/mem0-plugin/hooks/codex-hooks.json b/mem0-plugin/hooks/codex-hooks.json index 0676308a9..23e6dbd32 100644 --- a/mem0-plugin/hooks/codex-hooks.json +++ b/mem0-plugin/hooks/codex-hooks.json @@ -18,7 +18,7 @@ { "type": "command", "command": "${CODEX_PLUGIN_ROOT}/scripts/on_user_prompt.sh", - "statusMessage": "Searching mem0 memories...", + "statusMessage": "Checking memory relevance...", "timeout": 5 } ] diff --git a/mem0-plugin/hooks/cursor-hooks.json b/mem0-plugin/hooks/cursor-hooks.json index 442d3718e..124148bea 100644 --- a/mem0-plugin/hooks/cursor-hooks.json +++ b/mem0-plugin/hooks/cursor-hooks.json @@ -15,10 +15,6 @@ "preCompact": [ { "command": "${CURSOR_PLUGIN_ROOT}/scripts/on_pre_compact.sh" - }, - { - "command": "python3 ${CURSOR_PLUGIN_ROOT}/scripts/on_pre_compact.py", - "timeout": 30 } ], "stop": [ diff --git a/mem0-plugin/hooks/hooks.json b/mem0-plugin/hooks/hooks.json index d5b907427..2c98eb9d3 100644 --- a/mem0-plugin/hooks/hooks.json +++ b/mem0-plugin/hooks/hooks.json @@ -30,12 +30,6 @@ "type": "command", "command": "${CLAUDE_PLUGIN_ROOT}/scripts/on_pre_compact.sh", "statusMessage": "Preparing pre-compaction summary..." - }, - { - "type": "command", - "command": "python3 ${CLAUDE_PLUGIN_ROOT}/scripts/on_pre_compact.py", - "statusMessage": "Saving session state to mem0...", - "timeout": 30 } ] } @@ -57,7 +51,7 @@ { "type": "command", "command": "${CLAUDE_PLUGIN_ROOT}/scripts/on_user_prompt.sh", - "statusMessage": "Searching mem0 memories...", + "statusMessage": "Checking memory relevance...", "timeout": 5 } ] diff --git a/mem0-plugin/scripts/_identity.py b/mem0-plugin/scripts/_identity.py new file mode 100644 index 000000000..0725be58e --- /dev/null +++ b/mem0-plugin/scripts/_identity.py @@ -0,0 +1,59 @@ +"""Resolve mem0 user_id with deterministic priority. + +Resolution priority: + 1. MEM0_USER_ID env var (explicit override) + 2. ~/.mem0/identity.json cache (pinned to current MEM0_API_KEY fingerprint) + 3. Derived: "mem0-" + sha256(MEM0_API_KEY)[:12] + 4. Fallback: $USER, else "default" + +Same MEM0_API_KEY across machines yields the same user_id, which fixes +the "47 user buckets per account" symptom from running on multiple +laptops with different $USER values. +""" + +from __future__ import annotations + +import hashlib +import json +import os +from datetime import datetime, timezone + +_CACHE_PATH = os.path.expanduser("~/.mem0/identity.json") + + +def resolve_user_id() -> str: + explicit = os.environ.get("MEM0_USER_ID", "").strip() + if explicit: + return explicit + + api_key = os.environ.get("MEM0_API_KEY", "").strip() + if api_key: + digest = hashlib.sha256(api_key.encode("utf-8")).hexdigest() + fingerprint = digest[:8] + + try: + with open(_CACHE_PATH, "r") as f: + cached = json.load(f) + if cached.get("api_key_fingerprint") == fingerprint and cached.get("user_id"): + return cached["user_id"] + except (OSError, json.JSONDecodeError): + pass + + derived = "mem0-" + digest[:12] + try: + os.makedirs(os.path.dirname(_CACHE_PATH), exist_ok=True) + with open(_CACHE_PATH, "w") as f: + json.dump( + { + "user_id": derived, + "source": "api_key", + "api_key_fingerprint": fingerprint, + "resolved_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + }, + f, + ) + except OSError: + pass + return derived + + return os.environ.get("USER") or "default" diff --git a/mem0-plugin/scripts/_identity.sh b/mem0-plugin/scripts/_identity.sh new file mode 100644 index 000000000..3e0b62a17 --- /dev/null +++ b/mem0-plugin/scripts/_identity.sh @@ -0,0 +1,57 @@ +# Source this file. Sets MEM0_RESOLVED_USER_ID. +# +# Resolution priority: +# 1. MEM0_USER_ID env var (explicit override) +# 2. ~/.mem0/identity.json cache (pinned to current MEM0_API_KEY fingerprint) +# 3. Derived: "mem0-" + sha256(MEM0_API_KEY)[:12] +# 4. Fallback: $USER, else "default" +# +# Same MEM0_API_KEY across machines yields the same user_id, which fixes +# the "47 user buckets per account" symptom from running on multiple +# laptops with different $USER values. + +_mem0_sha256() { + if command -v sha256sum >/dev/null 2>&1; then + sha256sum | cut -d' ' -f1 + else + shasum -a 256 | cut -d' ' -f1 + fi +} + +_mem0_resolve_identity() { + if [ -n "${MEM0_USER_ID:-}" ]; then + printf '%s' "$MEM0_USER_ID" + return + fi + + local api_key="${MEM0_API_KEY:-}" + local cache="$HOME/.mem0/identity.json" + + if [ -n "$api_key" ]; then + local digest + digest=$(printf '%s' "$api_key" | _mem0_sha256) + local fp="${digest:0:8}" + + if [ -f "$cache" ]; then + local cached_fp cached_id + cached_fp=$(jq -r '.api_key_fingerprint // ""' "$cache" 2>/dev/null) + cached_id=$(jq -r '.user_id // ""' "$cache" 2>/dev/null) + if [ "$cached_fp" = "$fp" ] && [ -n "$cached_id" ]; then + printf '%s' "$cached_id" + return + fi + fi + + local derived="mem0-${digest:0:12}" + mkdir -p "$HOME/.mem0" 2>/dev/null && \ + printf '{"user_id":"%s","source":"api_key","api_key_fingerprint":"%s","resolved_at":"%s"}\n' \ + "$derived" "$fp" "$(date -u +%FT%TZ)" > "$cache" 2>/dev/null + printf '%s' "$derived" + return + fi + + printf '%s' "${USER:-default}" +} + +MEM0_RESOLVED_USER_ID="$(_mem0_resolve_identity)" +export MEM0_RESOLVED_USER_ID diff --git a/mem0-plugin/scripts/block_memory_write.sh b/mem0-plugin/scripts/block_memory_write.sh index 0cf6dbc84..bb38686d6 100755 --- a/mem0-plugin/scripts/block_memory_write.sh +++ b/mem0-plugin/scripts/block_memory_write.sh @@ -13,6 +13,10 @@ set -euo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + INPUT=$(cat) FILE_PATH=$(echo "$INPUT" | jq -r '.tool_input.file_path // .tool_input.path // ""' 2>/dev/null || echo "") @@ -22,7 +26,7 @@ if [ -z "$FILE_PATH" ]; then fi case "$FILE_PATH" in - */MEMORY.md|*/memory/*.md|*/.claude/*/memory/*) + */MEMORY.md|*/.claude/memory/*) echo "BLOCKED: Do not write to $FILE_PATH. Use the mem0 MCP \`add_memory\` tool instead to persist memories. This project uses mem0 for all memory storage." >&2 exit 2 ;; diff --git a/mem0-plugin/scripts/capture_compact_summary.py b/mem0-plugin/scripts/capture_compact_summary.py new file mode 100644 index 000000000..b6df7c42e --- /dev/null +++ b/mem0-plugin/scripts/capture_compact_summary.py @@ -0,0 +1,172 @@ +#!/usr/bin/env python3 +"""Capture the post-compaction summary into mem0. + +PreCompact hooks fire BEFORE the summary is generated, so they can't +store the actual compact-summary text. This script runs at +SessionStart with source=compact, reads the transcript, finds the +most recent entry flagged isCompactSummary=true, and stores it as a +memory tagged metadata.type=compact_summary. + +Input: JSON on stdin with transcript_path, session_id, source +Output: stderr logs only (exit 0 always -- must not block) + +Spawned in the background by on_session_start.sh; the user-facing +bootstrap text continues without waiting on the network. +""" + +from __future__ import annotations + +import json +import logging +import os +import sys +import urllib.error +import urllib.request +from datetime import date, timedelta + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from _identity import resolve_user_id + +log = logging.getLogger("mem0-compact-summary") +log.setLevel(logging.DEBUG) +_handler = logging.StreamHandler(sys.stderr) +_handler.setFormatter(logging.Formatter("[mem0-compact-summary] %(message)s")) +log.addHandler(_handler) + +if os.environ.get("MEM0_DEBUG"): + _log_dir = os.path.expanduser("~/.mem0") + try: + os.makedirs(_log_dir, exist_ok=True) + _file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) + _file_handler.setFormatter(logging.Formatter("[mem0-compact-summary] %(asctime)s %(message)s")) + log.addHandler(_file_handler) + except OSError: + pass + +API_URL = "https://api.mem0.ai" +MAX_TAIL_LINES = 2000 +MAX_SUMMARY_CHARS = 50000 +# Compact summaries describe a single session's state -- stale after a quarter. +COMPACT_SUMMARY_EXPIRY_DAYS = 90 + + +def tail_lines(filepath: str, n: int) -> list[str]: + try: + with open(filepath, "rb") as f: + f.seek(0, 2) + file_size = f.tell() + if file_size == 0: + return [] + chunk_size = min(file_size, n * 4096) + f.seek(max(0, file_size - chunk_size)) + data = f.read().decode("utf-8", errors="replace") + return data.splitlines()[-n:] + except OSError: + return [] + + +def find_compact_summary(lines: list[str]) -> str: + """Walk transcript backwards, return text content of the most recent + entry flagged isCompactSummary=true. Empty string if none found.""" + for line in reversed(lines): + line = line.strip() + if not line: + continue + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + if not entry.get("isCompactSummary"): + continue + + message = entry.get("message", {}) + content = message.get("content", []) + if isinstance(content, str): + return content[:MAX_SUMMARY_CHARS] + if isinstance(content, list): + parts = [] + for block in content: + if isinstance(block, str): + parts.append(block) + elif isinstance(block, dict) and block.get("type") == "text": + parts.append(block.get("text", "")) + return "\n".join(parts).strip()[:MAX_SUMMARY_CHARS] + return "" + + +def store_summary(api_key: str, summary: str, user_id: str, session_id: str) -> bool: + expires = (date.today() + timedelta(days=COMPACT_SUMMARY_EXPIRY_DAYS)).isoformat() + body = { + "messages": [{"role": "user", "content": summary}], + "user_id": user_id, + "metadata": { + "type": "compact_summary", + "source": "session-start-compact", + "session_id": session_id, + }, + "infer": False, + "expiration_date": expires, + } + + data = json.dumps(body).encode("utf-8") + req = urllib.request.Request( + f"{API_URL}/v1/memories/", + data=data, + headers={ + "Content-Type": "application/json", + "Authorization": f"Token {api_key}", + }, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=15) as resp: + if resp.status in (200, 201): + log.info("Compact summary stored") + return True + log.warning("API returned status %d", resp.status) + return False + except urllib.error.URLError as e: + log.warning("API call failed: %s", e) + return False + + +def main(): + api_key = os.environ.get("MEM0_API_KEY", "") + if not api_key: + log.debug("MEM0_API_KEY not set, skipping capture") + return + + try: + hook_input = json.loads(sys.stdin.read()) + except (json.JSONDecodeError, OSError): + log.debug("No valid JSON on stdin") + return + + transcript_path = hook_input.get("transcript_path", "") + if not transcript_path: + log.debug("No transcript_path provided") + return + + session_id = hook_input.get("session_id", "") + user_id = resolve_user_id() + + lines = tail_lines(transcript_path, MAX_TAIL_LINES) + if not lines: + log.debug("Transcript empty or unreadable: %s", transcript_path) + return + + summary = find_compact_summary(lines) + if not summary: + log.debug("No isCompactSummary entry found") + return + + log.info("Capturing compact summary (%d chars)", len(summary)) + store_summary(api_key, summary, user_id, session_id) + + +if __name__ == "__main__": + try: + main() + except Exception as e: + log.error("Unexpected error: %s", e) + sys.exit(0) diff --git a/mem0-plugin/scripts/on_pre_compact.py b/mem0-plugin/scripts/on_pre_compact.py index fb59e8d5c..6f280842e 100755 --- a/mem0-plugin/scripts/on_pre_compact.py +++ b/mem0-plugin/scripts/on_pre_compact.py @@ -18,8 +18,12 @@ import json import logging import os import sys -import urllib.request import urllib.error +import urllib.request +from datetime import date, timedelta + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from _identity import resolve_user_id log = logging.getLogger("mem0-capture") log.setLevel(logging.DEBUG) @@ -27,11 +31,25 @@ _handler = logging.StreamHandler(sys.stderr) _handler.setFormatter(logging.Formatter("[mem0-capture] %(message)s")) log.addHandler(_handler) +if os.environ.get("MEM0_DEBUG"): + _log_dir = os.path.expanduser("~/.mem0") + try: + os.makedirs(_log_dir, exist_ok=True) + _file_handler = logging.FileHandler(os.path.join(_log_dir, "hooks.log")) + _file_handler.setFormatter(logging.Formatter("[mem0-capture] %(asctime)s %(message)s")) + log.addHandler(_file_handler) + except OSError: + pass + API_URL = "https://api.mem0.ai" MAX_TAIL_LINES = 500 MAX_USER_MESSAGES = 30 MAX_BASH_COMMANDS = 20 MAX_ASSISTANT_TEXT = 10000 +# session_state captures churn fast (active codebase, files in flight). Past +# ~3 months they're stale noise. Durable facts (decisions, conventions) are +# stored separately by the agent without an expiration_date. +SESSION_STATE_EXPIRY_DAYS = 90 def tail_lines(filepath: str, n: int) -> list[str]: @@ -149,8 +167,9 @@ def build_content(state: dict, source: str) -> str: return "\n".join(parts) -def store_memory(api_key: str, content: str, user_id: str, source: str) -> bool: +def store_memory(api_key: str, content: str, user_id: str, source: str, session_id: str = "") -> bool: """Store session state as a memory via the Mem0 REST API.""" + expires = (date.today() + timedelta(days=SESSION_STATE_EXPIRY_DAYS)).isoformat() body = { "messages": [ {"role": "user", "content": content} @@ -159,7 +178,9 @@ def store_memory(api_key: str, content: str, user_id: str, source: str) -> bool: "metadata": { "type": "session_state", "source": source, + "session_id": session_id, }, + "expiration_date": expires, } data = json.dumps(body).encode("utf-8") @@ -207,7 +228,8 @@ def main(): log.debug("No transcript_path provided") return - user_id = os.environ.get("MEM0_USER_ID", os.environ.get("USER", "default")) + session_id = hook_input.get("session_id", "") + user_id = resolve_user_id() lines = tail_lines(transcript_path, MAX_TAIL_LINES) if not lines: @@ -228,7 +250,7 @@ def main(): len(state["bash_commands"]), ) - store_memory(api_key, content, user_id, source) + store_memory(api_key, content, user_id, source, session_id) if __name__ == "__main__": diff --git a/mem0-plugin/scripts/on_pre_compact.sh b/mem0-plugin/scripts/on_pre_compact.sh index 0cd1dead2..a3dffd5eb 100755 --- a/mem0-plugin/scripts/on_pre_compact.sh +++ b/mem0-plugin/scripts/on_pre_compact.sh @@ -5,12 +5,16 @@ # the full context before it gets compressed. # # Output: Text instructions injected into Claude's context. -# Claude still has the full conversation and can write an accurate summary. -# A companion Python script (on_pre_compact.py) also runs to capture -# transcript state directly via the Mem0 REST API as a safety net. +# Claude still has the full conversation and can write an accurate summary, +# which it stores via add_memory(infer=False) so the platform preserves +# the structure verbatim instead of running a second extraction pass. set -euo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + cat <<'EOF' ## CRITICAL: Pre-Compaction Session Summary @@ -18,7 +22,9 @@ Context compaction is about to happen. You are about to lose most of your conver ### Step 1: Store session summary -Call `add_memory` with a thorough summary covering ALL of the following: +Call `add_memory` with `infer=False` and a thorough summary covering ALL of the following. + +`infer=False` is critical here: you've already done the extraction work yourself using full context. Without it, the platform runs a second LLM pass that loses your structure and pulls fragmented facts. With it, your summary is preserved verbatim. ``` ## Session Summary (Pre-Compaction) @@ -44,11 +50,19 @@ Call `add_memory` with a thorough summary covering ALL of the following: the post-compaction agent continue without asking redundant questions] ``` -Include metadata: `{"type": "session_state", "source": "pre-compaction"}` +Tool call shape: +``` +add_memory( + messages=[{"role":"user","content":""}], + user_id="", + metadata={"type":"session_state","source":"pre-compaction"}, + infer=False, +) +``` ### Step 2: Store any unstored learnings -If there are learnings from this session that you haven't stored yet, store them as separate memories: +If there are learnings from this session that you haven't stored yet, store them as separate memories with `infer=False` (same reasoning -- you've already extracted the fact, don't re-extract): - Failed approaches -> metadata `{"type": "anti_pattern"}` - Successful strategies -> metadata `{"type": "task_learning"}` - Architecture decisions -> metadata `{"type": "decision"}` diff --git a/mem0-plugin/scripts/on_session_start.sh b/mem0-plugin/scripts/on_session_start.sh index 353a17130..cd31eb39a 100755 --- a/mem0-plugin/scripts/on_session_start.sh +++ b/mem0-plugin/scripts/on_session_start.sh @@ -11,9 +11,34 @@ # even if jq is missing or stdin is malformed. set -uo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + +# Skip the bootstrap entirely if no API key is configured -- the agent +# would otherwise be told to call mem0 MCP tools that will all fail. +if [ -z "${MEM0_API_KEY:-}" ]; then + exit 0 +fi + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +# shellcheck source=_identity.sh +. "$SCRIPT_DIR/_identity.sh" + INPUT=$(cat) SOURCE=$(echo "$INPUT" | jq -r '.source // "startup"' 2>/dev/null || echo "startup") +# Identity line is emitted before every bootstrap variant so the agent +# uses the same user_id the hooks resolved. Without this, the agent's +# search_memories/add_memory MCP calls may bind to a different bucket +# than what the hooks write to. +echo "## Mem0 Identity" +echo "" +echo "Active user_id: \`$MEM0_RESOLVED_USER_ID\`" +echo "" +echo "Always include \`{\"user_id\": \"$MEM0_RESOLVED_USER_ID\"}\` (wrapped in an \`AND\` clause) in every \`search_memories\` filter and as \`user_id\` on every \`add_memory\` call. This keeps memories under one bucket regardless of which machine you're on." +echo "" + if [ "$SOURCE" = "startup" ]; then cat <<'EOF' ## Mem0 Session Bootstrap @@ -40,14 +65,22 @@ Continue where you left off. EOF elif [ "$SOURCE" = "compact" ]; then + # Capture the just-generated compact summary in the background. + # PreCompact fires too early to see this entry; SessionStart-compact + # is the first place isCompactSummary=true is in the transcript. + echo "$INPUT" | python3 "$SCRIPT_DIR/capture_compact_summary.py" 2>/dev/null & + cat <<'EOF' ## Mem0 Post-Compaction Recovery -Context was just compacted. You may have lost important session context. +Context was just compacted. The Claude Code-generated compact summary +is being captured to mem0 in the background as `metadata.type=compact_summary`. -1. Call `search_memories` with queries related to what you were working on to reload relevant knowledge. -2. Check for any session state memories that were saved before compaction. -3. Continue working based on the recovered context. +1. Call `search_memories` to reload context, layering up to three angles: + - `metadata.type=session_state` -- the rich pre-compaction summary you wrote + - `metadata.type=compact_summary` -- the platform-generated condensed summary just now + - `metadata.type=decision` / `anti_pattern` -- specific facts you stored during the session +2. Continue working from the recovered context. EOF fi diff --git a/mem0-plugin/scripts/on_stop.sh b/mem0-plugin/scripts/on_stop.sh index 5dd088f69..848ab9906 100755 --- a/mem0-plugin/scripts/on_stop.sh +++ b/mem0-plugin/scripts/on_stop.sh @@ -12,6 +12,10 @@ set -euo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" INPUT=$(cat) diff --git a/mem0-plugin/scripts/on_stop_codex.sh b/mem0-plugin/scripts/on_stop_codex.sh index 80c16d311..de80b0fed 100755 --- a/mem0-plugin/scripts/on_stop_codex.sh +++ b/mem0-plugin/scripts/on_stop_codex.sh @@ -17,6 +17,10 @@ set -uo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + INPUT=$(cat) STOP_HOOK_ACTIVE=$(echo "$INPUT" | jq -r '.stop_hook_active // false' 2>/dev/null || echo "false") diff --git a/mem0-plugin/scripts/on_task_completed.sh b/mem0-plugin/scripts/on_task_completed.sh index befc1b34a..57a568027 100755 --- a/mem0-plugin/scripts/on_task_completed.sh +++ b/mem0-plugin/scripts/on_task_completed.sh @@ -9,6 +9,10 @@ set -euo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + INPUT=$(cat) TASK_SUBJECT=$(echo "$INPUT" | jq -r '.task_subject // "unknown task"' 2>/dev/null || echo "unknown task") diff --git a/mem0-plugin/scripts/on_user_prompt.sh b/mem0-plugin/scripts/on_user_prompt.sh index 398ec7456..50906a1d9 100755 --- a/mem0-plugin/scripts/on_user_prompt.sh +++ b/mem0-plugin/scripts/on_user_prompt.sh @@ -1,61 +1,72 @@ #!/usr/bin/env bash # Hook: UserPromptSubmit # -# Fires on every user message. Searches mem0 for relevant memories -# and injects them into Claude's context before processing. +# Fires on every user message. Instead of pre-searching mem0 with the +# raw prompt, this injects a decision rubric telling the agent when +# and how to search itself. The agent has more context than this +# script does -- let it decide. # -# Input: JSON on stdin with prompt, session_id, cwd, transcript_path -# Output: Matching memories as context text (exit 0) -# -# Skips search for very short prompts (< 20 chars) and when -# MEM0_API_KEY is not set. Uses a 3s timeout to minimize latency. +# Input: JSON on stdin (prompt, session_id, cwd, transcript_path) +# Output: Decision rubric injected into Claude's context (exit 0) -# Intentionally omit -e so the script always exits 0 even if -# curl or jq fail — must never block the user's prompt. +# Intentionally omit -e so the script always exits 0 even if jq fails -- +# must never block the user's prompt. set -uo pipefail +if [ -n "${MEM0_DEBUG:-}" ]; then + mkdir -p "$HOME/.mem0" && exec 2>>"$HOME/.mem0/hooks.log" +fi + INPUT=$(cat) PROMPT=$(echo "$INPUT" | jq -r '.prompt // ""' 2>/dev/null || echo "") -# Skip trivial prompts — not worth a network call +# Acknowledgements and short replies don't warrant memory context if [ ${#PROMPT} -lt 20 ]; then exit 0 fi -API_KEY="${MEM0_API_KEY:-}" -if [ -z "$API_KEY" ]; then +# No API key means the agent can't search anyway +if [ -z "${MEM0_API_KEY:-}" ]; then exit 0 fi -USER_ID="${MEM0_USER_ID:-${USER:-default}}" +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +# shellcheck source=_identity.sh +. "$SCRIPT_DIR/_identity.sh" +USER_ID="$MEM0_RESOLVED_USER_ID" -# Build request body safely via jq to avoid injection -BODY=$(jq -n --arg query "$PROMPT" --arg user_id "$USER_ID" \ - '{query: $query, filters: {user_id: $user_id}, top_k: 5}') +cat </dev/null || echo "") +Before responding, decide whether persistent memory context from mem0 would +improve your answer. The agent -- not this hook -- owns this decision. -if [ -z "$RESPONSE" ]; then - exit 0 -fi +**Search WHEN** the user: +- references past work, decisions, or things "we" built +- asks "how should we...", "best way to...", or any decision-style question +- hits an error, bug, or asks for debugging help +- requests work that touches their stack, tools, conventions, or preferences +- starts a non-trivial task in a known project -# Extract memories from response (API returns a flat array) -MEMORIES=$(echo "$RESPONSE" | jq -r ' - if type == "array" then . else .results // [] end | - if length == 0 then empty else - "## Relevant memories from mem0\n\n" + - (map(select(.memory != null) | "- " + .memory) | join("\n")) - end -' 2>/dev/null || echo "") +**Skip WHEN:** +- the prompt is an acknowledgement or continuation +- the user is *stating* new info -- that's a write trigger (\`add_memory\`), not a search +- it's a pure syntax / factual question answerable from general knowledge +- you already searched this scope earlier in the turn -if [ -n "$MEMORIES" ]; then - echo "$MEMORIES" -fi +**If searching, do it well:** +- Run **2-4 parallel** \`search_memories\` calls with different angles, not one + query that echoes the user's prompt. +- Phrase queries as **nouns** ("auth module decisions"), not full sentences. +- Filter shape: the root must be a logical operator (\`AND\` / \`OR\` / \`NOT\`) + with an array, and metadata uses a **nested** object (not dotted keys). + Combine \`user_id\` with one \`metadata.type\` clause per call: + - \`{"AND": [{"user_id": "$USER_ID"}, {"metadata": {"type": "decision"}}]}\` -- design / architecture + - \`{"AND": [{"user_id": "$USER_ID"}, {"metadata": {"type": "anti_pattern"}}]}\` -- debugging, error handling + - \`{"AND": [{"user_id": "$USER_ID"}, {"metadata": {"type": "user_preference"}}]}\` -- tooling, stack, style + - \`{"AND": [{"user_id": "$USER_ID"}, {"metadata": {"type": "convention"}}]}\` -- established patterns +- Or scope with just \`{"AND": [{"user_id": "$USER_ID"}]}\` when no metadata filter fits. +- Empty results are normal -- proceed without context. +EOF exit 0 diff --git a/mem0-plugin/scripts/setup_coding_categories.py b/mem0-plugin/scripts/setup_coding_categories.py new file mode 100644 index 000000000..57d35066e --- /dev/null +++ b/mem0-plugin/scripts/setup_coding_categories.py @@ -0,0 +1,142 @@ +#!/usr/bin/env python3 +"""Replace mem0's default category taxonomy with one tuned for coding workflows. + +mem0 auto-tags every memory with one or more `categories`. By default the list +is consumer-oriented (food, hobbies, music, ...), which is meaningless for code. +This script replaces the project's category list with a coding-focused one. + +The change is project-level (per the platform docs, per-request overrides are +not supported on the managed API). Run once per project; future memories will +be tagged using the new list automatically. + +Usage: + python setup_coding_categories.py # dry-run: show current vs proposed, no changes + python setup_coding_categories.py --apply # actually call project.update() + +Requires the mem0ai Python SDK and MEM0_API_KEY to be set. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys + +CODING_CATEGORIES = [ + { + "architecture_decisions": ( + "Design choices, system structure, technology selection, trade-offs evaluated, " + "and architectural patterns adopted in the project." + ) + }, + { + "anti_patterns": ( + "Approaches that failed, debugging dead-ends, common mistakes to avoid, " + "and lessons learned from things that didn't work." + ) + }, + { + "task_learnings": ( + "Strategies and approaches that succeeded for specific tasks, including tooling " + "tricks, workflow shortcuts, and effective problem-solving patterns." + ) + }, + { + "tooling_setup": ( + "Development environment, build tools, dependencies, package managers, deploy " + "pipelines, and configuration steps for the project." + ) + }, + { + "bug_fixes": ( + "Specific bug fixes with root cause analysis, the fix applied, and how the bug " + "was diagnosed -- useful for recognising similar issues later." + ) + }, + { + "coding_conventions": ( + "Code style, naming patterns, file organisation, error-handling conventions, " + "and team agreements about how code is written in this project." + ) + }, + { + "user_preferences": ( + "User's stated preferences for tools, libraries, languages, formatting, " + "and ways of working." + ) + }, +] + + +def _print_categories(label: str, cats): + print(f"=== {label} ===") + if cats: + print(json.dumps(cats, indent=2)) + else: + print("(none / using mem0 defaults)") + print() + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument( + "--apply", + action="store_true", + help="Actually call project.update(). Without this flag, runs in dry-run mode.", + ) + args = ap.parse_args() + + if not os.environ.get("MEM0_API_KEY"): + print("ERROR: MEM0_API_KEY is not set. Export it and try again.", file=sys.stderr) + return 1 + + try: + from mem0 import MemoryClient + except ImportError: + print( + "ERROR: the mem0ai Python SDK is not installed.\n" + "Install with: pip install mem0ai\n" + "Then re-run this script.", + file=sys.stderr, + ) + return 1 + + try: + client = MemoryClient() + except Exception as e: + print( + f"ERROR initialising MemoryClient: {e}\n" + "Most commonly this is an invalid MEM0_API_KEY -- check the key at " + "https://app.mem0.ai/dashboard/api-keys", + file=sys.stderr, + ) + return 1 + + try: + current = client.project.get(fields=["custom_categories"]) + current_cats = current.get("custom_categories") if isinstance(current, dict) else None + except Exception as e: + print(f"ERROR fetching current categories: {e}", file=sys.stderr) + return 1 + + _print_categories("Current project categories", current_cats) + _print_categories("Proposed coding categories", CODING_CATEGORIES) + + if not args.apply: + print("Dry-run only -- no changes made. Re-run with --apply to write.") + return 0 + + print("Applying coding categories...") + try: + response = client.project.update(custom_categories=CODING_CATEGORIES) + except Exception as e: + print(f"ERROR applying update: {e}", file=sys.stderr) + return 1 + + print("Done.", response if response else "") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/mem0-plugin/skills/mem0-codex/SKILL.md b/mem0-plugin/skills/mem0-codex/SKILL.md deleted file mode 100644 index 036f1a242..000000000 --- a/mem0-plugin/skills/mem0-codex/SKILL.md +++ /dev/null @@ -1,62 +0,0 @@ ---- -name: mem0-codex -description: > - Mem0 persistent memory integration for Codex. Automatically retrieve relevant - memories at the start of each task, store key learnings when tasks complete, - and capture session state before context is lost. Use the mem0 MCP tools - (add_memory, search_memories, get_memories, etc.) for all memory operations. ---- - -# Mem0 Memory Protocol for Codex - -You have access to persistent memory via the mem0 MCP tools. Follow this protocol to maintain context across sessions. - -## On every new task - -1. Call `search_memories` with a query related to the current task or project to load relevant context. -2. Review returned memories to understand what has been learned in prior sessions. -3. If appropriate, call `get_memories` to browse all stored memories for this user. - -## After completing significant work - -Extract key learnings and store them using the `add_memory` tool: - -- **Decisions made** -> Include metadata `{"type": "decision"}` -- **Strategies that worked** -> Include metadata `{"type": "task_learning"}` -- **Failed approaches** -> Include metadata `{"type": "anti_pattern"}` -- **User preferences observed** -> Include metadata `{"type": "user_preference"}` -- **Environment/setup discoveries** -> Include metadata `{"type": "environmental"}` -- **Conventions established** -> Include metadata `{"type": "convention"}` - -Memories can be as detailed as needed -- include full context, reasoning, code snippets, file paths, and examples. Longer, searchable memories are more valuable than vague one-liners. - -## Before losing context - -If context is about to be compacted or the session is ending, store a comprehensive session summary: - -``` -## Session Summary - -### User's Goal -[What the user originally asked for] - -### What Was Accomplished -[Numbered list of tasks completed] - -### Key Decisions Made -[Architectural choices, trade-offs discussed] - -### Files Created or Modified -[Important file paths with what changed] - -### Current State -[What is in progress, pending items, next steps] -``` - -Include metadata: `{"type": "session_state"}` - -## Memory hygiene - -- Do NOT write to MEMORY.md or any file-based memory. Use mem0 MCP tools exclusively. -- Only store genuinely useful learnings. Skip trivial interactions. -- Use specific, searchable language in memory content. diff --git a/mem0-plugin/skills/mem0-mcp/SKILL.md b/mem0-plugin/skills/mem0-mcp/SKILL.md new file mode 100644 index 000000000..bfdd15789 --- /dev/null +++ b/mem0-plugin/skills/mem0-mcp/SKILL.md @@ -0,0 +1,170 @@ +--- +name: mem0-mcp +description: > + Mem0 memory protocol for agents using the mem0 MCP tools (Claude Code, Cursor, + Codex, and any other MCP-aware runtime). Decide deliberately when memory context + would help, run targeted searches with metadata filters when it would, and store + key learnings as work completes. Use the mem0 MCP tools (add_memory, + search_memories, get_memories, etc.) for all memory operations. +--- + +# Mem0 MCP Memory Protocol + +You have access to persistent memory via the mem0 MCP tools. Follow this protocol to maintain context across sessions. + +## On every new task + +Decide whether persistent memory context would improve your response, then act accordingly. Don't search by default — search deliberately. + +### Decide: search or skip? + +**Search WHEN** the user: +- references past work, decisions, or things "we" built +- asks "how should we...", "best way to...", or any decision-style question +- hits an error, bug, or asks for debugging help +- requests work that touches their stack, tools, conventions, or preferences +- starts a non-trivial task in a known project + +**Skip WHEN:** +- the prompt is an acknowledgement or continuation ("ok", "thanks", "continue") +- the user is *stating* new info — that's a write trigger (`add_memory`), not a search +- it's a pure syntax / factual question answerable from general knowledge +- you already searched this scope earlier in the turn + +Empty results are normal. Proceed without context — they don't mean the system is broken. + +### How to search well + +When you do search, run **2–4 parallel** `search_memories` calls at different angles instead of one query echoing the user's prompt. + +**Query phrasing:** +- Use **nouns**, not sentences. `"auth module decisions"` beats `"what did we decide about auth"`. +- Strip conversational filler. *"remember when we picked Postgres?"* → search `"Postgres choice"`. +- Use entity names, not pronouns. Resolve "that thing" from recent context first. +- Don't search on meta-questions ("what was that?") — use recent context or `get_memories` ordered by `created_at`. + +**Metadata filters** match the same `type` values written under "After completing significant work" below. + +Two rules from the v2 filter spec: + +1. The root **must** be a logical operator (`AND` / `OR` / `NOT`) with an array. A bare `{"user_id": "..."}` won't work. +2. Metadata uses a **nested** object, not a dotted key. `{"metadata": {"type": "decision"}}`, never `{"metadata.type": "decision"}`. Only top-level metadata keys are filterable. + +Combine `user_id` with one metadata clause per call: + +| `metadata.type` clause | Use for | +|--------|---------| +| `{"metadata": {"type": "decision"}}` | design / architecture / "how should we" questions | +| `{"metadata": {"type": "anti_pattern"}}` | debugging, error handling, things that failed before | +| `{"metadata": {"type": "user_preference"}}` | tooling, stack, style — always include for code work | +| `{"metadata": {"type": "convention"}}` | established patterns in this project | + +Full filter (replace `` with the active user_id from your runtime): +```python +filters={"AND": [{"user_id": ""}, {"metadata": {"type": "decision"}}]} +``` + +### Worked example + +User asks: *"Refactor the auth module to use JWT."* + +Don't: +```python +search_memories(query="Refactor the auth module to use JWT") +# Hits whatever shares words. Misses prior decisions and preferences. +``` + +Do (parallel — substitute the active `user_id` for ``): +```python +search_memories(query="auth module decisions", + filters={"AND": [{"user_id": ""}, {"metadata": {"type": "decision"}}]}) +search_memories(query="JWT", + filters={"AND": [{"user_id": ""}]}) +search_memories(query="auth refactor failures", + filters={"AND": [{"user_id": ""}, {"metadata": {"type": "anti_pattern"}}]}) +search_memories(query="auth", + filters={"AND": [{"user_id": ""}, {"metadata": {"type": "user_preference"}}]}) +``` + +## After completing significant work + +Extract key learnings and store them using the `add_memory` tool: + +- **Decisions made** -> Include metadata `{"type": "decision"}` +- **Strategies that worked** -> Include metadata `{"type": "task_learning"}` +- **Failed approaches** -> Include metadata `{"type": "anti_pattern"}` +- **User preferences observed** -> Include metadata `{"type": "user_preference"}` +- **Environment/setup discoveries** -> Include metadata `{"type": "environmental"}` +- **Conventions established** -> Include metadata `{"type": "convention"}` + +> `metadata.type` (which you set explicitly) and `categories` (which the platform auto-tags after the project's custom-category list — see `scripts/setup_coding_categories.py`) are complementary. Always set `metadata.type` for explicit filtering; the platform fills in `categories` on its own. Don't try to set `categories` on `add_memory` calls — per-request overrides aren't supported on the managed API. + +### Expiration: high-churn vs durable + +Some memory types are state snapshots that go stale fast; others are durable facts that should outlive the session that created them. Mark the difference with `expiration_date` on writes. + +| Type | Expiration | Why | +|---|---|---| +| `session_state`, `compact_summary` | `expiration_date` ≈ today + 90 days | Describe a single moment of project state. Useless after a quarter; clutter the recall surface. | +| `decision`, `anti_pattern`, `convention`, `user_preference`, `task_learning`, `environmental` | omit `expiration_date` | Durable facts. A decision made last year is still a decision; same for a convention or a user preference. | + +`add_memory` accepts `expiration_date` as a string (`"YYYY-MM-DD"`). The two server-side hooks (`on_pre_compact.py`, `capture_compact_summary.py`) already set this for the types they write. When you write directly via the MCP tool, follow the same rule. + +### Recency filter on recall + +When the user is asking about *current* state ("where were we", "what's the active task", "the latest decision on X"), filter recall to recent memories so stale snapshots don't surface: + +```python +# Last 90 days only +{"AND": [{"user_id": ""}, {"metadata": {"type": "session_state"}}, {"created_at": {"gte": "<90 days ago, YYYY-MM-DD>"}}]} +``` + +Skip the recency filter when the user is asking about durable facts ("what conventions does this project use", "have we hit this bug before") — those are timeless and recency would hide them. + +Memories can be as detailed as needed -- include full context, reasoning, code snippets, file paths, and examples. Longer, searchable memories are more valuable than vague one-liners. + +### Use `infer=False` for already-structured content + +When you've done the extraction work yourself — pre-compaction summaries, decisions, anti-patterns, conventions you've explicitly identified — pass `infer=False` so the platform stores your text verbatim instead of running a second extraction pass over it. + +```python +add_memory( + messages=[{"role": "user", "content": ""}], + user_id="", + metadata={"type": "decision"}, + infer=False, +) +``` + +Stick to one mode per distinct piece of content — don't mix `infer=True` (default) and `infer=False` for the same fact, you'll get duplicates. Default (`infer=True`) is right for raw conversational signal you want extracted; `infer=False` is right for pre-extracted structure. + +## Before losing context + +If context is about to be compacted or the session is ending, store a comprehensive session summary: + +``` +## Session Summary + +### User's Goal +[What the user originally asked for] + +### What Was Accomplished +[Numbered list of tasks completed] + +### Key Decisions Made +[Architectural choices, trade-offs discussed] + +### Files Created or Modified +[Important file paths with what changed] + +### Current State +[What is in progress, pending items, next steps] +``` + +Include metadata: `{"type": "session_state"}` + +## Memory hygiene + +- Do NOT write to MEMORY.md or any file-based memory. Use mem0 MCP tools exclusively. +- Only store genuinely useful learnings. Skip trivial interactions. +- Use specific, searchable language in memory content. diff --git a/mem0-ts/package.json b/mem0-ts/package.json index ec7235bac..b83f33c4f 100644 --- a/mem0-ts/package.json +++ b/mem0-ts/package.json @@ -1,6 +1,6 @@ { "name": "mem0ai", - "version": "3.0.2", + "version": "3.0.3", "description": "The Memory Layer For Your AI Apps", "main": "./dist/index.js", "module": "./dist/index.mjs", diff --git a/mem0-ts/src/client/config.ts b/mem0-ts/src/client/config.ts new file mode 100644 index 000000000..646360661 --- /dev/null +++ b/mem0-ts/src/client/config.ts @@ -0,0 +1,165 @@ +/** + * Best-effort read/write of ~/.mem0/config.json from the TS SDK. + * + * Used to stitch PostHog identities: SDKs and CLIs persist anonymous + * distinct_id values here, and the TS MemoryClient reads those on init to + * fire $identify and merge them into the email identity. + * + * Node-only. Browsers (no `process.versions.node`) no-op. + */ + +export interface Mem0AnonIds { + oss?: string; + cli?: string; + aliasedPairs: string[]; +} + +interface NodeFs { + fs: typeof import("fs"); + path: typeof import("path"); + crypto: typeof import("crypto"); + configPath: string; +} + +async function getNodeFs(): Promise { + if (typeof process === "undefined" || !process.versions?.node) return null; + try { + const [fs, path, os, crypto] = await Promise.all([ + import("fs"), + import("path"), + import("os"), + import("crypto"), + ]); + const fsMod = (fs as any).default ?? fs; + const pathMod = (path as any).default ?? path; + const osMod = (os as any).default ?? os; + const cryptoMod = (crypto as any).default ?? crypto; + const dir = process.env.MEM0_DIR || pathMod.join(osMod.homedir(), ".mem0"); + return { + fs: fsMod, + path: pathMod, + crypto: cryptoMod, + configPath: pathMod.join(dir, "config.json"), + }; + } catch { + return null; + } +} + +function loadConfig(node: NodeFs): Record | null { + try { + if (!node.fs.existsSync(node.configPath)) return null; + const parsed = JSON.parse(node.fs.readFileSync(node.configPath, "utf8")); + return parsed && typeof parsed === "object" ? parsed : null; + } catch { + return null; + } +} + +function writeConfig(node: NodeFs, config: Record): void { + node.fs.mkdirSync(node.path.dirname(node.configPath), { recursive: true }); + node.fs.writeFileSync(node.configPath, JSON.stringify(config, null, 4)); +} + +function aliasPairMarker(node: NodeFs, anonId: string, email: string): string { + return node.crypto + .createHash("sha256") + .update(`${anonId}\0${email}`, "utf8") + .digest("hex"); +} + +function randomUserId(node: NodeFs): string { + if (typeof node.crypto.randomUUID === "function") { + return node.crypto.randomUUID(); + } + return ( + Math.random().toString(36).substring(2, 15) + + Math.random().toString(36).substring(2, 15) + ); +} + +export async function getOrCreateMem0UserId(): Promise { + const node = await getNodeFs(); + if (!node) return null; + try { + const config = loadConfig(node) ?? {}; + if (typeof config.user_id === "string" && config.user_id) { + return config.user_id; + } + const userId = randomUserId(node); + config.user_id = userId; + writeConfig(node, config); + return userId; + } catch { + return null; + } +} + +export async function readMem0AnonIds(): Promise { + const node = await getNodeFs(); + if (!node) return null; + const config = loadConfig(node); + if (!config) return null; + const telemetry = + config.telemetry && typeof config.telemetry === "object" + ? config.telemetry + : {}; + return { + oss: typeof config.user_id === "string" ? config.user_id : undefined, + cli: + typeof telemetry.anonymous_id === "string" + ? telemetry.anonymous_id + : undefined, + aliasedPairs: Array.isArray(telemetry.aliased_pairs) + ? telemetry.aliased_pairs.filter( + (item: unknown) => typeof item === "string", + ) + : [], + }; +} + +export async function isMem0Aliased( + anonId: string, + email: string, +): Promise { + if (!anonId || !email) return false; + const node = await getNodeFs(); + if (!node) return false; + const config = loadConfig(node); + if (!config) return false; + const telemetry = + config.telemetry && typeof config.telemetry === "object" + ? config.telemetry + : {}; + const aliasedPairs = Array.isArray(telemetry.aliased_pairs) + ? telemetry.aliased_pairs + : []; + return aliasedPairs.includes(aliasPairMarker(node, anonId, email)); +} + +export async function markMem0Aliased( + anonId: string, + email: string, +): Promise { + const node = await getNodeFs(); + if (!node) return; + try { + const config = loadConfig(node) ?? {}; + const telemetry = + config.telemetry && typeof config.telemetry === "object" + ? config.telemetry + : {}; + const aliasedPairs = Array.isArray(telemetry.aliased_pairs) + ? telemetry.aliased_pairs + : []; + const marker = aliasPairMarker(node, anonId, email); + if (!aliasedPairs.includes(marker)) { + aliasedPairs.push(marker); + } + telemetry.aliased_pairs = aliasedPairs; + config.telemetry = telemetry; + writeConfig(node, config); + } catch { + // Best-effort: read-only filesystems and unwritable paths just skip. + } +} diff --git a/mem0-ts/src/client/mem0.ts b/mem0-ts/src/client/mem0.ts index 783d7f925..cac2fe36c 100644 --- a/mem0-ts/src/client/mem0.ts +++ b/mem0-ts/src/client/mem0.ts @@ -20,7 +20,18 @@ import { CreateMemoryExportPayload, GetMemoryExportPayload, } from "./mem0.types"; -import { captureClientEvent, generateHash } from "./telemetry"; +import { + captureClientEvent, + generateHash, + isTelemetryEnabled, + telemetry, +} from "./telemetry"; +import { + getOrCreateMem0UserId, + isMem0Aliased, + markMem0Aliased, + readMem0AnonIds, +} from "./config"; import { camelToSnake, camelToSnakeKeys, snakeToCamelKeys } from "./utils"; import { createExceptionFromResponse, MemoryError } from "../common/exceptions"; @@ -118,6 +129,8 @@ export default class MemoryClient { this.telemetryId = generateHash(this.apiKey); } + await this._maybeAliasAnonToEmail(); + captureClientEvent("init", this, { client_type: "MemoryClient", }).catch((error: any) => { @@ -132,6 +145,30 @@ export default class MemoryClient { } } + private async _maybeAliasAnonToEmail(): Promise { + if (!isTelemetryEnabled()) return; + try { + const email = this.telemetryId; + if (!email || !email.includes("@")) return; + const sharedAnonId = await getOrCreateMem0UserId(); + const anonIds = await readMem0AnonIds(); + if (!anonIds && !sharedAnonId) return; + const candidates = [anonIds?.oss || sharedAnonId, anonIds?.cli].filter( + (id): id is string => !!id && id !== email, + ); + const seen = new Set(); + for (const anonId of candidates) { + if (seen.has(anonId) || (await isMem0Aliased(anonId, email))) continue; + seen.add(anonId); + if (await telemetry.captureIdentify(anonId, email)) { + await markMem0Aliased(anonId, email); + } + } + } catch (error: any) { + console.error("Failed to alias telemetry identity:", error); + } + } + private _captureEvent(methodName: string, args: any[]) { captureClientEvent(methodName, this, { success: true, diff --git a/mem0-ts/src/client/mem0.types.ts b/mem0-ts/src/client/mem0.types.ts index 1dacce376..afd5b663d 100644 --- a/mem0-ts/src/client/mem0.types.ts +++ b/mem0-ts/src/client/mem0.types.ts @@ -50,6 +50,13 @@ export interface PromptUpdatePayload { memoryDepth?: string | null; usecaseSetting?: string | number; multilingual?: boolean; + /** + * Toggle Memory Decay for this project. When `true`, search-time ranking + * boosts recently-used memories and gently dampens stale ones; when `false`, + * ranking is restored to the pre-decay behaviour. Off by default. + * See https://docs.mem0.ai/platform/features/memory-decay + */ + decay?: boolean; [key: string]: any; } diff --git a/mem0-ts/src/client/telemetry.ts b/mem0-ts/src/client/telemetry.ts index ab5887820..df3a722d4 100644 --- a/mem0-ts/src/client/telemetry.ts +++ b/mem0-ts/src/client/telemetry.ts @@ -32,8 +32,12 @@ class UnifiedTelemetry implements TelemetryClient { this.host = host; } - async captureEvent(distinctId: string, eventName: string, properties = {}) { - if (!MEM0_TELEMETRY) return; + async captureEvent( + distinctId: string, + eventName: string, + properties = {}, + ): Promise { + if (!MEM0_TELEMETRY) return false; const eventProperties = { client_version: version, @@ -61,9 +65,50 @@ class UnifiedTelemetry implements TelemetryClient { if (!response.ok) { console.error("Telemetry event capture failed:", await response.text()); + return false; } + return true; } catch (error) { console.error("Telemetry event capture failed:", error); + return false; + } + } + + async captureIdentify(anonId: string, email: string): Promise { + if (!MEM0_TELEMETRY) return false; + if (!anonId || !email || anonId === email) return false; + + const payload = { + api_key: this.apiKey, + distinct_id: email, + event: "$identify", + properties: { + $anon_distinct_id: anonId, + client_source: "typescript", + $lib: "posthog-node", + }, + }; + + try { + const response = await fetch(this.host, { + method: "POST", + headers: { + "Content-Type": "application/json", + }, + body: JSON.stringify(payload), + }); + + if (!response.ok) { + console.error( + "Telemetry identify capture failed:", + await response.text(), + ); + return false; + } + return true; + } catch (error) { + console.error("Telemetry identify capture failed:", error); + return false; } } @@ -72,6 +117,10 @@ class UnifiedTelemetry implements TelemetryClient { } } +function isTelemetryEnabled(): boolean { + return MEM0_TELEMETRY; +} + const telemetry = new UnifiedTelemetry(POSTHOG_API_KEY, POSTHOG_HOST); async function captureClientEvent( @@ -101,4 +150,4 @@ async function captureClientEvent( ); } -export { telemetry, captureClientEvent, generateHash }; +export { telemetry, captureClientEvent, generateHash, isTelemetryEnabled }; diff --git a/mem0-ts/src/client/telemetry.types.ts b/mem0-ts/src/client/telemetry.types.ts index 5b307d99d..cca3a58f4 100644 --- a/mem0-ts/src/client/telemetry.types.ts +++ b/mem0-ts/src/client/telemetry.types.ts @@ -3,7 +3,7 @@ export interface TelemetryClient { distinctId: string, eventName: string, properties?: Record, - ): Promise; + ): Promise; shutdown(): Promise; } diff --git a/mem0-ts/src/client/tests/telemetry-aliasing.test.ts b/mem0-ts/src/client/tests/telemetry-aliasing.test.ts new file mode 100644 index 000000000..a0351bbf1 --- /dev/null +++ b/mem0-ts/src/client/tests/telemetry-aliasing.test.ts @@ -0,0 +1,410 @@ +/** + * Tests for PostHog identity stitching in the TS MemoryClient. + * + * Covers $identify firing, idempotency via pair markers, and the node/browser + * gate. Mocks fs and fetch; never touches the real ~/.mem0/config.json. + */ +import * as fs from "fs"; +import * as os from "os"; +import * as path from "path"; +import { MemoryClient } from "../mem0"; +import { telemetry } from "../telemetry"; +import { + getOrCreateMem0UserId, + isMem0Aliased, + markMem0Aliased, + readMem0AnonIds, +} from "../config"; +import { TEST_API_KEY } from "./helpers"; +import { setupMockFetch, installConsoleSuppression } from "./setup"; + +installConsoleSuppression(); + +function setupMockFetchWithPostHog(): jest.Mock { + return setupMockFetch( + new Map([["us.i.posthog.com", { status: 200, body: "ok" }]]), + ); +} + +// ─── config.ts (node-only fs read/write) ────────────────────── + +describe("config.ts — readMem0AnonIds / markMem0Aliased", () => { + let tmpHome: string; + const originalMem0Dir = process.env.MEM0_DIR; + + beforeEach(() => { + tmpHome = fs.mkdtempSync(path.join(os.tmpdir(), "mem0-ts-test-")); + process.env.MEM0_DIR = tmpHome; + }); + + afterEach(() => { + if (fs.existsSync(tmpHome)) { + fs.rmSync(tmpHome, { recursive: true, force: true }); + } + if (originalMem0Dir === undefined) { + delete process.env.MEM0_DIR; + } else { + process.env.MEM0_DIR = originalMem0Dir; + } + }); + + test("returns null when config file does not exist", async () => { + expect(await readMem0AnonIds()).toBeNull(); + }); + + test("reads OSS user_id only", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ user_id: "oss-uuid" }), + ); + const ids = await readMem0AnonIds(); + expect(ids).toEqual({ + oss: "oss-uuid", + cli: undefined, + aliasedPairs: [], + }); + }); + + test("reads CLI anonymous_id and aliased_pairs", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ + telemetry: { anonymous_id: "cli-anon", aliased_pairs: ["pair-marker"] }, + }), + ); + const ids = await readMem0AnonIds(); + expect(ids).toEqual({ + oss: undefined, + cli: "cli-anon", + aliasedPairs: ["pair-marker"], + }); + }); + + test("getOrCreateMem0UserId creates and reuses shared SDK user_id", async () => { + const first = await getOrCreateMem0UserId(); + const second = await getOrCreateMem0UserId(); + expect(first).toBeTruthy(); + expect(second).toBe(first); + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.user_id).toBe(first); + }); + + test("returns null on malformed JSON", async () => { + fs.writeFileSync(path.join(tmpHome, "config.json"), "{not json"); + expect(await readMem0AnonIds()).toBeNull(); + }); + + test("markMem0Aliased preserves other fields", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ + user_id: "oss-uuid", + telemetry: { anonymous_id: "cli-anon" }, + }), + ); + await markMem0Aliased("oss-uuid", "user@example.com"); + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.user_id).toBe("oss-uuid"); + expect(written.telemetry.anonymous_id).toBe("cli-anon"); + expect(written.telemetry.aliased_pairs).toHaveLength(1); + expect(await isMem0Aliased("oss-uuid", "user@example.com")).toBe(true); + }); + + test("markMem0Aliased creates telemetry section when missing", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ user_id: "oss-uuid" }), + ); + await markMem0Aliased("oss-uuid", "user@example.com"); + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.telemetry.aliased_pairs).toHaveLength(1); + }); + + test("markMem0Aliased tracks each pair independently", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ user_id: "oss-uuid" }), + ); + await markMem0Aliased("oss-uuid", "user@example.com"); + expect(await isMem0Aliased("oss-uuid", "user@example.com")).toBe(true); + expect(await isMem0Aliased("other-uuid", "user@example.com")).toBe(false); + expect(await isMem0Aliased("oss-uuid", "other@example.com")).toBe(false); + }); + + test("markMem0Aliased does not throw when target dir is unwritable", async () => { + // Point at a path that cannot be written to (a file-as-dir collision). + fs.writeFileSync(path.join(tmpHome, "blocker"), "x"); + process.env.MEM0_DIR = path.join(tmpHome, "blocker"); // file used as dir + await expect( + markMem0Aliased("oss-uuid", "user@example.com"), + ).resolves.toBeUndefined(); + }); +}); + +// ─── telemetry.captureIdentify ─────────────────────────────── + +describe("telemetry.captureIdentify", () => { + test("fires $identify with $anon_distinct_id", async () => { + const fetchMock = jest.fn(async () => ({ + ok: true, + status: 200, + text: async () => "ok", + })) as unknown as typeof fetch; + global.fetch = fetchMock as any; + + await telemetry.captureIdentify("anon-uuid", "user@example.com"); + + expect(fetchMock).toHaveBeenCalledTimes(1); + const [, init] = (fetchMock as jest.Mock).mock.calls[0]; + const payload = JSON.parse(init.body); + expect(payload.event).toBe("$identify"); + expect(payload.distinct_id).toBe("user@example.com"); + expect(payload.properties.$anon_distinct_id).toBe("anon-uuid"); + expect(payload.properties.$process_person_profile).toBeUndefined(); + }); + + test("skips when anon equals email", async () => { + const fetchMock = jest.fn() as unknown as typeof fetch; + global.fetch = fetchMock as any; + await telemetry.captureIdentify("user@example.com", "user@example.com"); + expect(fetchMock).not.toHaveBeenCalled(); + }); + + test("skips when either input is empty", async () => { + const fetchMock = jest.fn() as unknown as typeof fetch; + global.fetch = fetchMock as any; + await telemetry.captureIdentify("", "user@example.com"); + await telemetry.captureIdentify("anon", ""); + expect(fetchMock).not.toHaveBeenCalled(); + }); +}); + +// ─── MemoryClient init aliasing ────────────────────────────── + +describe("MemoryClient — _maybeAliasAnonToEmail", () => { + let tmpHome: string; + const originalMem0Dir = process.env.MEM0_DIR; + + beforeEach(() => { + tmpHome = fs.mkdtempSync(path.join(os.tmpdir(), "mem0-ts-init-")); + process.env.MEM0_DIR = tmpHome; + }); + + afterEach(() => { + if (fs.existsSync(tmpHome)) { + fs.rmSync(tmpHome, { recursive: true, force: true }); + } + if (originalMem0Dir === undefined) { + delete process.env.MEM0_DIR; + } else { + process.env.MEM0_DIR = originalMem0Dir; + } + }); + + // Construct a non-initialised client so we can call _maybeAliasAnonToEmail + // in isolation (the real constructor's _initializeClient also fires it). + function makeStubClient(telemetryId: string): MemoryClient { + const client = Object.create(MemoryClient.prototype) as MemoryClient; + (client as any).apiKey = TEST_API_KEY; + (client as any).host = "https://api.mem0.ai"; + (client as any).telemetryId = telemetryId; + return client; + } + + test("fires $identify on first init and persists pair marker", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ user_id: "oss-uuid" }), + ); + const fetchMock = setupMockFetchWithPostHog(); + + const client = makeStubClient("test@example.com"); + await (client as any)._maybeAliasAnonToEmail(); + + const identifyCalls = (fetchMock.mock.calls as any[]).filter( + ([, init]: [string, RequestInit]) => { + if (!init?.body) return false; + return JSON.parse(init.body as string).event === "$identify"; + }, + ); + expect(identifyCalls.length).toBe(1); + const body = JSON.parse(identifyCalls[0][1].body); + expect(body.distinct_id).toBe("test@example.com"); + expect(body.properties.$anon_distinct_id).toBe("oss-uuid"); + + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.telemetry.aliased_pairs).toHaveLength(1); + }); + + test("platform-first init creates shared anon ID and identifies it", async () => { + const fetchMock = setupMockFetchWithPostHog(); + + const client = makeStubClient("test@example.com"); + await (client as any)._maybeAliasAnonToEmail(); + + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.user_id).toBeTruthy(); + expect(written.telemetry.aliased_pairs).toHaveLength(1); + + const identifyCalls = (fetchMock.mock.calls as any[]).filter( + ([, init]: [string, RequestInit]) => { + if (!init?.body) return false; + return JSON.parse(init.body as string).event === "$identify"; + }, + ); + expect(identifyCalls.length).toBe(1); + const body = JSON.parse(identifyCalls[0][1].body); + expect(body.distinct_id).toBe("test@example.com"); + expect(body.properties.$anon_distinct_id).toBe(written.user_id); + }); + + test("second init does not refire $identify", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ + user_id: "oss-uuid", + telemetry: {}, + }), + ); + await markMem0Aliased("oss-uuid", "test@example.com"); + const fetchMock = setupMockFetchWithPostHog(); + + const client = makeStubClient("test@example.com"); + await (client as any)._maybeAliasAnonToEmail(); + + const identifyCalls = (fetchMock.mock.calls as any[]).filter( + ([, init]: [string, RequestInit]) => { + if (!init?.body) return false; + return JSON.parse(init.body as string).event === "$identify"; + }, + ); + expect(identifyCalls.length).toBe(0); + }); + + test("fires $identify for both OSS and CLI anon ids", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ + user_id: "oss-uuid", + telemetry: { anonymous_id: "cli-anon" }, + }), + ); + const fetchMock = setupMockFetchWithPostHog(); + + const client = makeStubClient("test@example.com"); + await (client as any)._maybeAliasAnonToEmail(); + + const identifyCalls = (fetchMock.mock.calls as any[]).filter( + ([, init]: [string, RequestInit]) => { + if (!init?.body) return false; + return JSON.parse(init.body as string).event === "$identify"; + }, + ); + expect(identifyCalls.length).toBe(2); + const anonIds = identifyCalls.map( + (c: [string, RequestInit]) => + JSON.parse(c[1].body as string).properties.$anon_distinct_id, + ); + expect(anonIds).toContain("oss-uuid"); + expect(anonIds).toContain("cli-anon"); + + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.telemetry.aliased_pairs).toHaveLength(2); + }); + + test("noop when telemetryId is not an email", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ user_id: "oss-uuid" }), + ); + const fetchMock = setupMockFetch(); + + const client = makeStubClient("not-an-email"); + await (client as any)._maybeAliasAnonToEmail(); + + const identifyCalls = (fetchMock.mock.calls as any[]).filter( + ([, init]: [string, RequestInit]) => { + if (!init?.body) return false; + return JSON.parse(init.body as string).event === "$identify"; + }, + ); + expect(identifyCalls.length).toBe(0); + }); + + test("does not throw when config read fails", async () => { + fs.writeFileSync(path.join(tmpHome, "config.json"), "{not json"); + setupMockFetch(); + + const client = makeStubClient("test@example.com"); + await expect( + (client as any)._maybeAliasAnonToEmail(), + ).resolves.toBeUndefined(); + }); + + test("noop when telemetry disabled — no fs read, no fs write, no events", async () => { + fs.writeFileSync( + path.join(tmpHome, "config.json"), + JSON.stringify({ user_id: "oss-uuid" }), + ); + const fetchMock = setupMockFetch(); + + jest.resetModules(); + const original = process.env.MEM0_TELEMETRY; + process.env.MEM0_TELEMETRY = "false"; + try { + const { MemoryClient: ColdClient } = await import("../mem0"); + const client = Object.create(ColdClient.prototype); + client.apiKey = TEST_API_KEY; + client.host = "https://api.mem0.ai"; + client.telemetryId = "test@example.com"; + await client._maybeAliasAnonToEmail(); + } finally { + if (original === undefined) delete process.env.MEM0_TELEMETRY; + else process.env.MEM0_TELEMETRY = original; + jest.resetModules(); + } + + const identifyCalls = (fetchMock.mock.calls as any[]).filter( + ([, init]: [string, RequestInit]) => { + if (!init?.body) return false; + return JSON.parse(init.body as string).event === "$identify"; + }, + ); + expect(identifyCalls.length).toBe(0); + + const written = JSON.parse( + fs.readFileSync(path.join(tmpHome, "config.json"), "utf8"), + ); + expect(written.telemetry?.aliased_pairs).toBeUndefined(); + }); +}); + +// ─── Browser env path (no process.versions.node) ───────────── + +describe("config.ts in browser-like environment", () => { + test("readMem0AnonIds returns null when not Node", async () => { + const originalProcess = global.process; + // @ts-expect-error force-undefining global to simulate a browser + delete global.process; + try { + jest.resetModules(); + const { readMem0AnonIds: browserRead } = await import("../config"); + expect(await browserRead()).toBeNull(); + } finally { + global.process = originalProcess; + jest.resetModules(); + } + }); +}); diff --git a/mem0-ts/src/oss/src/memory/index.ts b/mem0-ts/src/oss/src/memory/index.ts index f2856c493..e175e71ff 100644 --- a/mem0-ts/src/oss/src/memory/index.ts +++ b/mem0-ts/src/oss/src/memory/index.ts @@ -53,6 +53,7 @@ import { ScoredResult, } from "../utils/scoring"; import { getDefaultVectorStoreDbPath } from "../utils/sqlite"; +import { getOrCreateMem0UserId } from "../../../client/config"; // Entity params that must be passed via filters - check both snake_case and camelCase const ENTITY_PARAMS = [ @@ -466,7 +467,12 @@ export class Memory { this.telemetryId === "anonymous" || this.telemetryId === "anonymous-supabase" ) { - this.telemetryId = await this.vectorStore.getUserId(); + this.telemetryId = + (await getOrCreateMem0UserId()) || + (await this.vectorStore.getUserId()); + try { + await this.vectorStore.setUserId(this.telemetryId); + } catch {} } return this.telemetryId; } catch (error) { diff --git a/mem0/client/main.py b/mem0/client/main.py index 34db1c782..69ff77bd8 100644 --- a/mem0/client/main.py +++ b/mem0/client/main.py @@ -19,8 +19,8 @@ from mem0.client.types import ( from mem0.client.utils import api_error_handler # Exception classes are referenced in docstrings only -from mem0.memory.setup import get_user_id, setup_config -from mem0.memory.telemetry import capture_client_event +from mem0.memory.setup import get_user_id, is_aliased, mark_aliased, read_anon_ids, setup_config +from mem0.memory.telemetry import capture_client_event, client_telemetry logger = logging.getLogger(__name__) @@ -33,6 +33,32 @@ setup_config() ENTITY_PARAMS = frozenset({"user_id", "agent_id", "app_id", "run_id"}) +def _maybe_alias_anon_to_email(user_email): + """Fire $identify per prior anon ID so PostHog merges them into email. + + Idempotent via telemetry.aliased_pairs: only writes markers when + telemetry is actually enabled, so disabling/re-enabling MEM0_TELEMETRY still works. + Best-effort: never raises. + """ + if client_telemetry.posthog is None: + return + if not user_email or "@" not in user_email: + return + try: + anon_ids = read_anon_ids() + seen = set() + for anon_id in (anon_ids.get("oss"), anon_ids.get("cli")): + if not anon_id or anon_id == user_email or anon_id in seen: + continue + seen.add(anon_id) + if is_aliased(anon_id, user_email): + continue + if client_telemetry.capture_identify(anon_id, user_email): + mark_aliased(anon_id, user_email) + except Exception as e: + logger.debug("Failed to alias anon telemetry to %r: %s", user_email, e) + + class MemoryClient: """Client for interacting with the Mem0 API. @@ -108,6 +134,7 @@ class MemoryClient: user_email=self.user_email, ) + _maybe_alias_anon_to_email(self.user_email) capture_client_event("client.init", self, {"sync_type": "sync"}) def _validate_api_key(self): @@ -985,6 +1012,7 @@ class AsyncMemoryClient: user_email=self.user_email, ) + _maybe_alias_anon_to_email(self.user_email) capture_client_event("client.init", self, {"sync_type": "async"}) def _validate_api_key(self): diff --git a/mem0/client/project.py b/mem0/client/project.py index fe466b7ef..4255bec18 100644 --- a/mem0/client/project.py +++ b/mem0/client/project.py @@ -398,6 +398,7 @@ class Project(BaseProject): custom_categories: Optional[List[str]] = None, retrieval_criteria: Optional[List[Dict[str, Any]]] = None, multilingual: Optional[bool] = None, + decay: Optional[bool] = None, ) -> Dict[str, Any]: """ Update project settings. @@ -407,6 +408,9 @@ class Project(BaseProject): custom_categories: New categories for the project retrieval_criteria: New retrieval criteria for the project multilingual: Whether to use the input language for memory storage and retrieval + decay: Toggle Memory Decay for this project. When True, search-time + ranking boosts recently-used memories and gently dampens stale ones; when + False, ranking is restored to the pre-decay behaviour. Off by default. Returns: Dictionary containing the API response. @@ -423,11 +427,12 @@ class Project(BaseProject): and custom_categories is None and retrieval_criteria is None and multilingual is None + and decay is None ): raise ValueError( "At least one parameter must be provided for update: " "custom_instructions, custom_categories, retrieval_criteria, " - "multilingual" + "multilingual, decay" ) payload = self._prepare_params( @@ -436,6 +441,7 @@ class Project(BaseProject): "custom_categories": custom_categories, "retrieval_criteria": retrieval_criteria, "multilingual": multilingual, + "decay": decay, } ) response = self._client.patch( @@ -451,6 +457,7 @@ class Project(BaseProject): "custom_categories": custom_categories, "retrieval_criteria": retrieval_criteria, "multilingual": multilingual, + "decay": decay, "sync_type": "sync", }, ) @@ -715,6 +722,7 @@ class AsyncProject(BaseProject): custom_categories: Optional[List[str]] = None, retrieval_criteria: Optional[List[Dict[str, Any]]] = None, multilingual: Optional[bool] = None, + decay: Optional[bool] = None, ) -> Dict[str, Any]: """ Update project settings. @@ -724,6 +732,9 @@ class AsyncProject(BaseProject): custom_categories: New categories for the project retrieval_criteria: New retrieval criteria for the project multilingual: Whether to use the input language for memory storage and retrieval + decay: Toggle Memory Decay for this project. When True, search-time + ranking boosts recently-used memories and gently dampens stale ones; when + False, ranking is restored to the pre-decay behaviour. Off by default. Returns: Dictionary containing the API response. @@ -740,11 +751,12 @@ class AsyncProject(BaseProject): and custom_categories is None and retrieval_criteria is None and multilingual is None + and decay is None ): raise ValueError( "At least one parameter must be provided for update: " "custom_instructions, custom_categories, retrieval_criteria, " - "multilingual" + "multilingual, decay" ) payload = self._prepare_params( @@ -753,6 +765,7 @@ class AsyncProject(BaseProject): "custom_categories": custom_categories, "retrieval_criteria": retrieval_criteria, "multilingual": multilingual, + "decay": decay, } ) response = await self._client.patch( @@ -768,6 +781,7 @@ class AsyncProject(BaseProject): "custom_categories": custom_categories, "retrieval_criteria": retrieval_criteria, "multilingual": multilingual, + "decay": decay, "sync_type": "async", }, ) diff --git a/mem0/memory/setup.py b/mem0/memory/setup.py index 3dbd70667..7d98bcae4 100644 --- a/mem0/memory/setup.py +++ b/mem0/memory/setup.py @@ -1,6 +1,8 @@ import json +import logging import os import uuid +from hashlib import sha256 # Set up the directory path VECTOR_ID = str(uuid.uuid4()) @@ -8,28 +10,113 @@ home_dir = os.path.expanduser("~") mem0_dir = os.environ.get("MEM0_DIR") or os.path.join(home_dir, ".mem0") os.makedirs(mem0_dir, exist_ok=True) +_logger = logging.getLogger(__name__) + + +def _config_path(): + return os.path.join(mem0_dir, "config.json") + + +def _load_config(): + """Load ~/.mem0/config.json, returning {} on missing/malformed file.""" + path = _config_path() + if not os.path.exists(path): + return {} + try: + with open(path, "r") as f: + data = json.load(f) + return data if isinstance(data, dict) else {} + except Exception as e: + _logger.debug("Failed to load mem0 config %s: %s", path, e) + return {} + + +def _write_config(config): + """Best-effort write of ~/.mem0/config.json. Never raises.""" + path = _config_path() + try: + with open(path, "w") as f: + json.dump(config, f, indent=4) + except Exception as e: + _logger.debug("Failed to write mem0 config %s: %s", path, e) + def setup_config(): - config_path = os.path.join(mem0_dir, "config.json") - if not os.path.exists(config_path): - user_id = str(uuid.uuid4()) - config = {"user_id": user_id} - with open(config_path, "w") as config_file: - json.dump(config, config_file, indent=4) + """Ensure ~/.mem0/config.json exists with a top-level user_id. + + Idempotent: backfills user_id for users whose config was written by the + CLI (which writes telemetry.anonymous_id but no top-level user_id). + Without this, OSS Python telemetry is silently dropped because + get_user_id() returns None when user_id is missing. + """ + config = _load_config() + if config.get("user_id"): + return + config["user_id"] = str(uuid.uuid4()) + _write_config(config) def get_user_id(): - config_path = os.path.join(mem0_dir, "config.json") - if not os.path.exists(config_path): + config = _load_config() + if not config: return "anonymous_user" + return config.get("user_id") - try: - with open(config_path, "r") as config_file: - config = json.load(config_file) - user_id = config.get("user_id") - return user_id - except Exception: - return "anonymous_user" + +def read_anon_ids(): + """Return anon IDs and alias markers from ~/.mem0/config.json. + + Returns a dict with keys "oss", "cli", "aliased_pairs" (IDs may be + None). OSS Python writes top-level "user_id"; the CLI writes + "telemetry.anonymous_id". They may coexist depending on which surface ran + first. + """ + config = _load_config() + telemetry = config.get("telemetry") if isinstance(config.get("telemetry"), dict) else {} + aliased_pairs = telemetry.get("aliased_pairs") + return { + "oss": config.get("user_id"), + "cli": telemetry.get("anonymous_id"), + "aliased_pairs": aliased_pairs if isinstance(aliased_pairs, list) else [], + } + + +def _alias_pair_marker(anon_id, email): + return sha256(f"{anon_id}\0{email}".encode("utf-8")).hexdigest() + + +def is_aliased(anon_id, email): + """Return whether anon_id -> email has already been identified.""" + if not anon_id or not email: + return False + config = _load_config() + telemetry = config.get("telemetry") if isinstance(config.get("telemetry"), dict) else {} + aliased_pairs = telemetry.get("aliased_pairs") + if not isinstance(aliased_pairs, list): + return False + return _alias_pair_marker(anon_id, email) in aliased_pairs + + +def mark_aliased(anon_id, email): + """Persist an anon_id -> email alias marker so $identify fires once per pair. + + The marker is hashed to avoid storing platform emails in the local config. + """ + if not anon_id or not email: + return + config = _load_config() + telemetry = config.get("telemetry") + if not isinstance(telemetry, dict): + telemetry = {} + aliased_pairs = telemetry.get("aliased_pairs") + if not isinstance(aliased_pairs, list): + aliased_pairs = [] + marker = _alias_pair_marker(anon_id, email) + if marker not in aliased_pairs: + aliased_pairs.append(marker) + telemetry["aliased_pairs"] = aliased_pairs + config["telemetry"] = telemetry + _write_config(config) def get_or_create_user_id(vector_store=None): diff --git a/mem0/memory/telemetry.py b/mem0/memory/telemetry.py index f0c46cbb2..0c2c544ad 100644 --- a/mem0/memory/telemetry.py +++ b/mem0/memory/telemetry.py @@ -48,7 +48,8 @@ MEM0_TELEMETRY_SAMPLE_RATE = _parse_sample_rate(os.environ.get("MEM0_TELEMETRY_S # Events that bypass sampling and always fire. Keep this set in sync with the # event names passed to capture_event() in mem0/memory/main.py. -_LIFECYCLE_EVENTS = frozenset({"mem0.init", "mem0.reset", "mem0._create_procedural_memory"}) +# $identify is included so PostHog person-merging is never lost to sampling. +_LIFECYCLE_EVENTS = frozenset({"mem0.init", "mem0.reset", "mem0._create_procedural_memory", "$identify"}) def _sampling_before_send(msg): @@ -112,6 +113,23 @@ class AnonymousTelemetry: except Exception as e: _logger.debug("Failed to capture telemetry event %r: %s", event_name, e) + def capture_identify(self, anon_id, email): + """Fire $identify with $anon_distinct_id so PostHog merges anon_id into email.""" + if self.posthog is None: + return False + if not anon_id or not email or anon_id == email: + return False + try: + self.posthog.capture( + distinct_id=email, + event="$identify", + properties={"$anon_distinct_id": anon_id, "client_source": "python"}, + ) + return True + except Exception as e: + _logger.debug("Failed to capture $identify for %r: %s", email, e) + return False + def close(self): if self.posthog is not None: self.posthog.shutdown() diff --git a/pyproject.toml b/pyproject.toml index 94bdf4e1a..998a81890 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "mem0ai" -version = "2.0.1" +version = "2.0.2" description = "Long-term memory for AI Agents" authors = [ { name = "Mem0", email = "support@mem0.ai" } @@ -153,3 +153,6 @@ known-first-party = ["mem0", "mem0_cli"] [tool.isort] profile = "black" known_first_party = ["mem0", "mem0_cli"] +# isort scope kept aligned with [tool.ruff.lint.isort] above. +# black-equivalent profile here matches the formatter behaviour ruff applies. +# Plugin-version bumps need a touch here to fire required CI checks (path-filter trap). diff --git a/scripts/oss-to-platform-migrate.sh b/scripts/oss-to-platform-migrate.sh new file mode 100755 index 000000000..b3a984e82 --- /dev/null +++ b/scripts/oss-to-platform-migrate.sh @@ -0,0 +1,1195 @@ +#!/usr/bin/env bash +set -euo pipefail + +if ! command -v python3 >/dev/null 2>&1; then + printf '%s\n' "Error: python3 is required to run the Mem0 migration. Install Python 3, then rerun this command." >&2 + exit 1 +fi + +exec python3 - "$@" <<'PY' +from __future__ import annotations + +import argparse +from collections import Counter +import getpass +import json +import os +import platform +import re +import stat +import sys +import uuid +from datetime import datetime, timezone +from hashlib import sha256 +from pathlib import Path +from typing import Any, Dict, List, Optional, Set, Tuple +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen + + +DEFAULT_BASE_URL = "https://api.mem0.ai" +POSTHOG_API_KEY = "phc_hgJkUVJFYtmaJqrvf6CYN67TIQ8yhXAkWzUn9AMU4yX" +POSTHOG_CAPTURE_URL = "https://us.i.posthog.com/i/v0/e/" +SCRIPT_VERSION = "oss-to-platform-migrate-v1" +EMAIL_RE = re.compile(r"^[^@\s]+@[^@\s]+\.[^@\s]+$") +DEFAULT_QDRANT_COLLECTION = "mem0" +DEFAULT_EXPORT_PAGE_SIZE = 100 +DEFAULT_PLATFORM_PAGE_SIZE = 100 + + +class MigrationError(Exception): + pass + + +class AuthError(MigrationError): + pass + + +class HTTPStatusError(MigrationError): + def __init__(self, status_code: int, detail: str): + self.status_code = status_code + self.detail = detail + super().__init__(detail) + + +def utc_now() -> str: + return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") + + +def print_info(message: str) -> None: + print(message) + + +def print_error(message: str) -> None: + print(f"Error: {message}", file=sys.stderr) + + +def mem0_dir() -> Path: + configured = os.environ.get("MEM0_DIR") + if configured: + return Path(configured).expanduser() + return Path.home() / ".mem0" + + +def config_path() -> Path: + return mem0_dir() / "config.json" + + +def load_config(path: Path) -> Dict[str, Any]: + if not path.exists(): + return {} + try: + data = json.loads(path.read_text(encoding="utf-8")) + return data if isinstance(data, dict) else {} + except Exception: + return {} + + +def write_config(path: Path, config: Dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + os.chmod(path.parent, stat.S_IRWXU) + path.write_text(json.dumps(config, indent=4) + "\n", encoding="utf-8") + os.chmod(path, stat.S_IRUSR | stat.S_IWUSR) + + +def ensure_oss_anon_id(path: Path, config: Dict[str, Any]) -> str: + existing = config.get("user_id") + if isinstance(existing, str) and existing: + return existing + + generated = str(uuid.uuid4()) + config["user_id"] = generated + try: + write_config(path, config) + except Exception: + # Keep the run usable even on read-only home directories. Telemetry can + # still use the transient ID for this run. + pass + return generated + + +def read_anon_ids(config: Dict[str, Any], fallback_oss_id: str) -> Dict[str, Any]: + telemetry = config.get("telemetry") + if not isinstance(telemetry, dict): + telemetry = {} + aliased_pairs = telemetry.get("aliased_pairs") + if not isinstance(aliased_pairs, list): + aliased_pairs = [] + return { + "oss": config.get("user_id") if isinstance(config.get("user_id"), str) else fallback_oss_id, + "cli": telemetry.get("anonymous_id") if isinstance(telemetry.get("anonymous_id"), str) else None, + "aliased_pairs": [item for item in aliased_pairs if isinstance(item, str)], + } + + +def primary_anon_id(anon_ids: Dict[str, Any]) -> str: + return anon_ids.get("oss") or anon_ids.get("cli") or "anonymous_user" + + +def unique_anon_ids(anon_ids: Dict[str, Any]) -> List[str]: + seen: Set[str] = set() + result: List[str] = [] + for anon_id in (anon_ids.get("oss"), anon_ids.get("cli")): + if not anon_id or anon_id in seen: + continue + seen.add(anon_id) + result.append(anon_id) + return result + + +def alias_pair_marker(anon_id: str, email: str) -> str: + return sha256(f"{anon_id}\0{email}".encode("utf-8")).hexdigest() + + +def is_aliased(config: Dict[str, Any], anon_id: str, email: str) -> bool: + telemetry = config.get("telemetry") + if not isinstance(telemetry, dict): + return False + aliased_pairs = telemetry.get("aliased_pairs") + if not isinstance(aliased_pairs, list): + return False + return alias_pair_marker(anon_id, email) in aliased_pairs + + +def mark_aliased(path: Path, config: Dict[str, Any], anon_id: str, email: str) -> None: + telemetry = config.get("telemetry") + if not isinstance(telemetry, dict): + telemetry = {} + aliased_pairs = telemetry.get("aliased_pairs") + if not isinstance(aliased_pairs, list): + aliased_pairs = [] + marker = alias_pair_marker(anon_id, email) + if marker not in aliased_pairs: + aliased_pairs.append(marker) + telemetry["aliased_pairs"] = aliased_pairs + config["telemetry"] = telemetry + write_config(path, config) + + +def normalize_email(email: str) -> str: + normalized = email.strip().lower() + if not EMAIL_RE.match(normalized): + raise MigrationError(f"Invalid email address: {email!r}") + return normalized + + +def parse_error_body(raw: bytes, fallback: str) -> str: + text = raw.decode("utf-8", errors="replace").strip() + if not text: + return fallback + try: + parsed = json.loads(text) + if isinstance(parsed, dict): + for key in ("error", "detail", "message"): + value = parsed.get(key) + if value: + return str(value) + except Exception: + pass + return text + + +def request_json( + method: str, + url: str, + *, + payload: Optional[Dict[str, Any]] = None, + headers: Optional[Dict[str, str]] = None, + timeout: float = 30.0, +) -> Dict[str, Any]: + request_headers = {"Accept": "application/json"} + if headers: + request_headers.update(headers) + + data = None + if payload is not None: + data = json.dumps(payload).encode("utf-8") + request_headers["Content-Type"] = "application/json" + + req = Request(url, data=data, headers=request_headers, method=method) + try: + with urlopen(req, timeout=timeout) as resp: + raw = resp.read() + except HTTPError as exc: + detail = parse_error_body(exc.read(), exc.reason or f"HTTP {exc.code}") + raise HTTPStatusError(exc.code, detail) from exc + except URLError as exc: + raise MigrationError(f"Network request failed for {url}: {exc.reason}. Check your connection and rerun the command.") from exc + + if not raw: + return {} + try: + parsed = json.loads(raw.decode("utf-8")) + except json.JSONDecodeError as exc: + raise MigrationError(f"Expected JSON response from {url}") from exc + return parsed if isinstance(parsed, dict) else {} + + +def post_json_best_effort(url: str, payload: Dict[str, Any], timeout: float = 5.0) -> bool: + data = json.dumps(payload).encode("utf-8") + req = Request(url, data=data, headers={"Content-Type": "application/json"}, method="POST") + try: + with urlopen(req, timeout=timeout) as resp: + resp.read() + return 200 <= resp.status < 300 + except Exception: + return False + + +def telemetry_enabled() -> bool: + raw = os.environ.get("MEM0_TELEMETRY", "true") + return raw.lower() in ("true", "1", "yes") + + +def telemetry_url() -> str: + return os.environ.get("MEM0_MIGRATE_TELEMETRY_URL", POSTHOG_CAPTURE_URL) + + +def base_event_properties(anon_ids: Dict[str, Any], email: Optional[str] = None) -> Dict[str, Any]: + properties: Dict[str, Any] = { + "client_source": "python", + "client_version": SCRIPT_VERSION, + "migration_phase": "migration", + "local_anonymous_id": primary_anon_id(anon_ids), + "oss_anonymous_id": anon_ids.get("oss"), + "cli_anonymous_id": anon_ids.get("cli"), + "python_version": sys.version, + "os": sys.platform, + "os_version": platform.version(), + "os_release": platform.release(), + "machine": platform.machine(), + "timestamp": utc_now(), + "$lib": "posthog-python", + } + if email: + properties["authenticated_email"] = email + return properties + + +def capture_migration_event( + event_name: str, + anon_ids: Dict[str, Any], + *, + email: Optional[str] = None, + additional: Optional[Dict[str, Any]] = None, +) -> bool: + if not telemetry_enabled(): + return False + distinct_id = email or primary_anon_id(anon_ids) + properties = base_event_properties(anon_ids, email) + if additional: + properties.update(additional) + payload = { + "api_key": POSTHOG_API_KEY, + "distinct_id": distinct_id, + "event": event_name, + "properties": properties, + } + return post_json_best_effort(telemetry_url(), payload) + + +def capture_identify(anon_id: str, email: str) -> bool: + if not telemetry_enabled() or not anon_id or not email or anon_id == email: + return False + payload = { + "api_key": POSTHOG_API_KEY, + "distinct_id": email, + "event": "$identify", + "properties": { + "$anon_distinct_id": anon_id, + "client_source": "python", + "client_version": SCRIPT_VERSION, + "$lib": "posthog-python", + }, + } + return post_json_best_effort(telemetry_url(), payload) + + +def stitch_identities(config_path_value: Path, config: Dict[str, Any], anon_ids: Dict[str, Any], email: str) -> None: + for anon_id in unique_anon_ids(anon_ids): + if anon_id == email or is_aliased(config, anon_id, email): + continue + if capture_identify(anon_id, email): + try: + mark_aliased(config_path_value, config, anon_id, email) + except Exception: + pass + + +def auth_headers(language: str = "python") -> Dict[str, str]: + return { + "X-Mem0-Source": "migration", + "X-Mem0-Client-Language": language, + "X-Mem0-Client-Version": SCRIPT_VERSION, + } + + +def platform_headers(api_key: str, language: str = "python") -> Dict[str, str]: + return { + "Authorization": f"Token {api_key}", + "X-Mem0-Source": "migration", + "X-Mem0-Client-Language": language, + "X-Mem0-Client-Version": SCRIPT_VERSION, + } + + +def api_base_url(args: argparse.Namespace, config: Dict[str, Any]) -> str: + platform_config = config.get("platform") if isinstance(config.get("platform"), dict) else {} + value = args.base_url or os.environ.get("MEM0_BASE_URL") or platform_config.get("base_url") or DEFAULT_BASE_URL + return str(value).rstrip("/") + + +def configured_api_key(args: argparse.Namespace, config: Dict[str, Any]) -> Tuple[Optional[str], str]: + if args.api_key: + return args.api_key, "flag" + env_key = os.environ.get("MEM0_API_KEY") + if env_key: + return env_key, "env" + platform_config = config.get("platform") if isinstance(config.get("platform"), dict) else {} + file_key = platform_config.get("api_key") + if isinstance(file_key, str) and file_key: + return file_key, "config" + return None, "" + + +def validate_api_key(api_key: str, base_url: str) -> str: + try: + result = request_json( + "GET", + f"{base_url}/v1/ping/", + headers={"Authorization": f"Token {api_key}"}, + timeout=5.0, + ) + except HTTPStatusError as exc: + if exc.status_code == 401: + raise AuthError("Invalid or expired API key.") from exc + raise + + email = result.get("user_email") + if not isinstance(email, str) or not email: + raise MigrationError("Platform authenticated the API key but did not return an account email.") + return normalize_email(email) + + +def prompt_line(prompt: str) -> str: + try: + with open("/dev/tty", "r", encoding="utf-8") as tty_in, open("/dev/tty", "w", encoding="utf-8") as tty_out: + tty_out.write(prompt) + tty_out.flush() + line = tty_in.readline() + except OSError as exc: + raise MigrationError( + "No interactive terminal is available. Re-run with --email and --code, or use --yes with a valid API key." + ) from exc + + if line == "": + raise MigrationError("No input received from terminal.") + return line.strip() + + +def prompt_line_default(prompt: str, default: str) -> str: + answer = prompt_line(f"{prompt} [{default}]: ") + return answer or default + + +def prompt_secret(prompt: str) -> str: + ensure_interactive_terminal() + try: + with open("/dev/tty", "w", encoding="utf-8") as tty_out: + return getpass.getpass(prompt, stream=tty_out).strip() + except OSError as exc: + raise MigrationError( + "No interactive terminal is available. Re-run with the required flags or environment variables." + ) from exc + + +def has_interactive_terminal() -> bool: + try: + with open("/dev/tty", "r", encoding="utf-8"), open("/dev/tty", "w", encoding="utf-8"): + return True + except OSError: + return False + + +def ensure_interactive_terminal() -> None: + try: + with open("/dev/tty", "r", encoding="utf-8"), open("/dev/tty", "w", encoding="utf-8"): + return + except OSError as exc: + raise MigrationError( + "No interactive terminal is available. Re-run with --email and --code, or use --yes with a valid API key." + ) from exc + + +def prompt_yes_no(prompt: str, *, default: bool = True) -> bool: + suffix = " [Y/n]: " if default else " [y/N]: " + answer = prompt_line(prompt + suffix).strip().lower() + if not answer: + return default + return answer in ("y", "yes") + + +def request_email_code(email: str, base_url: str) -> None: + try: + request_json( + "POST", + f"{base_url}/api/v1/auth/email_code/", + payload={"email": email}, + headers=auth_headers(), + ) + except HTTPStatusError as exc: + if exc.status_code == 429: + raise MigrationError("Too many attempts. Try again in a few minutes.") from exc + raise MigrationError(f"Failed to send verification code: {exc.detail}") from exc + + +def verify_email_code(email: str, code: str, base_url: str) -> str: + try: + result = request_json( + "POST", + f"{base_url}/api/v1/auth/email_code/verify/", + payload={"email": email, "code": code.strip()}, + headers=auth_headers(), + ) + except HTTPStatusError as exc: + if exc.status_code == 429: + raise MigrationError("Too many attempts. Try again in a few minutes.") from exc + raise MigrationError(f"Verification failed: {exc.detail}") from exc + + api_key = result.get("api_key") + if not isinstance(api_key, str) or not api_key: + raise MigrationError("Auth succeeded but no API key was returned. Contact support.") + return api_key + + +def email_code_auth(args: argparse.Namespace, base_url: str) -> Tuple[str, str, str]: + email = normalize_email(args.email) if args.email else normalize_email(prompt_line("Email: ")) + code = args.code + + if not code: + ensure_interactive_terminal() + request_email_code(email, base_url) + print_info("Verification code sent. Check your email.") + code = prompt_line("Verification code: ") + if not code: + raise MigrationError("Verification code is required.") + + api_key = verify_email_code(email, code, base_url) + resolved_email = validate_api_key(api_key, base_url) + return api_key, resolved_email, "email_code" + + +def existing_session_auth(args: argparse.Namespace, api_key: str, key_source: str, base_url: str) -> Optional[Tuple[str, str, str]]: + try: + email = validate_api_key(api_key, base_url) + except AuthError: + if key_source in ("flag", "env"): + raise + print_info("Stored Mem0 Platform API key is invalid or expired. Continuing with email login.") + return None + except MigrationError: + if key_source in ("flag", "env"): + raise + print_info("Could not validate stored Mem0 Platform session. Continuing with email login.") + return None + + if args.email: + return None + + if not args.yes: + print_info("") + print_info("Found an existing Mem0 Platform session.") + print_info("") + print_info(f"Account: {email}") + print_info("This migration will copy your local OSS memories into this Platform account.") + print_info("") + if not prompt_yes_no(f"Continue with {email}?", default=True): + return None + + return api_key, email, "existing_api_key" + + +def resolve_auth(args: argparse.Namespace, config: Dict[str, Any], base_url: str) -> Tuple[str, str, str]: + if args.api_key and args.email: + raise MigrationError("Cannot use both --api-key and --email.") + if args.code and not args.email: + raise MigrationError("--code requires --email.") + + api_key, key_source = configured_api_key(args, config) + if api_key: + result = existing_session_auth(args, api_key, key_source, base_url) + if result: + return result + + return email_code_auth(args, base_url) + + +def default_export_path() -> Path: + timestamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + return mem0_dir() / "migrations" / f"mem0-qdrant-export-{timestamp}.json" + + +def strip_env_assignment(value: str, name: str) -> str: + prefix = f"{name}=" + return value[len(prefix):] if value.startswith(prefix) else value + + +def qdrant_url(args: argparse.Namespace) -> str: + value = args.qdrant_url + env_value = os.environ.get("QDRANT_URL") + if not value and env_value: + if has_interactive_terminal(): + normalized_env_value = strip_env_assignment(env_value.strip(), "QDRANT_URL") + print_info("") + print_info("Found QDRANT_URL in your environment:") + print_info(normalized_env_value) + if prompt_yes_no("Use this Qdrant URL?", default=True): + value = env_value + else: + value = env_value + if not value: + if not has_interactive_terminal(): + raise MigrationError("Hosted Qdrant export requires --qdrant-url or QDRANT_URL.") + print_info("") + print_info("Python OSS hosted Qdrant export setup") + print_info("Enter the Qdrant Cloud details used by your OSS Memory configuration.") + value = prompt_line("Qdrant URL: ") + value = strip_env_assignment(value.strip(), "QDRANT_URL") + if not value: + raise MigrationError("Qdrant URL is required.") + return value.rstrip("/") + + +def qdrant_api_key(args: argparse.Namespace) -> str: + value = args.qdrant_api_key + env_value = os.environ.get("QDRANT_API_KEY") + if not value and env_value: + if has_interactive_terminal(): + print_info("") + print_info("Found QDRANT_API_KEY in your environment.") + if prompt_yes_no("Use this Qdrant API key?", default=True): + value = env_value + else: + value = env_value + if not value: + if not has_interactive_terminal(): + raise MigrationError("Hosted Qdrant export requires --qdrant-api-key or QDRANT_API_KEY.") + value = prompt_secret("Qdrant API key: ") + value = strip_env_assignment(value.strip(), "QDRANT_API_KEY") + if not value: + raise MigrationError("Qdrant API key is required.") + return value + + +def qdrant_collection(args: argparse.Namespace) -> str: + value = args.qdrant_collection or os.environ.get("QDRANT_COLLECTION") + if not value: + if has_interactive_terminal(): + value = prompt_line_default("Qdrant collection", DEFAULT_QDRANT_COLLECTION) + else: + value = DEFAULT_QDRANT_COLLECTION + value = value.strip() + if not value: + raise MigrationError("Qdrant collection name cannot be empty.") + return value + + +def qdrant_headers(api_key: str) -> Dict[str, str]: + return {"api-key": api_key} + + +def qdrant_request_json( + method: str, + url: str, + *, + api_key: str, + payload: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + try: + return request_json(method, url, payload=payload, headers=qdrant_headers(api_key), timeout=30.0) + except HTTPStatusError as exc: + if exc.status_code in (401, 403): + raise MigrationError("Qdrant authentication failed. Check your QDRANT_API_KEY.") from exc + if exc.status_code == 404: + raise MigrationError(f"Qdrant resource not found: {url}") from exc + raise MigrationError(f"Qdrant request failed: {exc.detail}") from exc + + +def preflight_qdrant_collection(base_url: str, api_key: str, collection_name: str) -> None: + qdrant_request_json( + "GET", + f"{base_url}/collections/{collection_name}", + api_key=api_key, + ) + + +def qdrant_filter(args: argparse.Namespace) -> Optional[Dict[str, Any]]: + filters = [] + for key, value in (("user_id", args.user_id), ("agent_id", args.agent_id), ("run_id", args.run_id)): + if value: + filters.append({"key": key, "match": {"value": value}}) + + if not filters: + if args.all: + return None + if not has_interactive_terminal(): + raise MigrationError("Export requires --user-id, --agent-id, --run-id, or --all.") + print_info("") + print_info("Choose which OSS memories to export.") + print_info("Enter a user_id to export one user, or press Enter to export the full collection.") + user_id = prompt_line("User ID to export (press Enter for full collection): ") + if user_id: + args.user_id = user_id + filters.append({"key": "user_id", "match": {"value": user_id}}) + else: + args.all = True + return None + + return {"must": filters} + + +def scroll_qdrant_points( + base_url: str, + api_key: str, + collection_name: str, + scroll_filter: Optional[Dict[str, Any]], + page_size: int, +) -> List[Dict[str, Any]]: + points: List[Dict[str, Any]] = [] + offset = None + + while True: + payload: Dict[str, Any] = { + "limit": page_size, + "with_payload": True, + "with_vector": False, + } + if scroll_filter is not None: + payload["filter"] = scroll_filter + if offset is not None: + payload["offset"] = offset + + response = qdrant_request_json( + "POST", + f"{base_url}/collections/{collection_name}/points/scroll", + api_key=api_key, + payload=payload, + ) + result = response.get("result") + if not isinstance(result, dict): + raise MigrationError("Qdrant scroll response did not include a result object.") + + page_points = result.get("points") + if not isinstance(page_points, list): + raise MigrationError("Qdrant scroll response did not include a points list.") + points.extend(point for point in page_points if isinstance(point, dict)) + + offset = result.get("next_page_offset") + if offset is None: + break + + return points + + +def normalize_qdrant_point(point: Dict[str, Any]) -> Dict[str, Any]: + payload = point.get("payload") + if not isinstance(payload, dict): + payload = {} + + promoted_payload_keys = ("user_id", "agent_id", "run_id", "actor_id", "role") + core_and_promoted_keys = { + "data", + "hash", + "created_at", + "updated_at", + "id", + "text_lemmatized", + "attributed_to", + *promoted_payload_keys, + } + + record: Dict[str, Any] = { + "id": str(point.get("id")), + "memory": payload.get("data", ""), + "hash": payload.get("hash"), + "created_at": payload.get("created_at"), + "updated_at": payload.get("updated_at"), + "source": "python_oss_qdrant", + "source_payload": payload, + } + + for key in promoted_payload_keys: + if key in payload: + record[key] = payload[key] + + metadata = {key: value for key, value in payload.items() if key not in core_and_promoted_keys} + if metadata: + record["metadata"] = metadata + + return record + + +def write_export_file(output_path: Path, artifact: Dict[str, Any]) -> None: + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(json.dumps(artifact, indent=2) + "\n", encoding="utf-8") + os.chmod(output_path, stat.S_IRUSR | stat.S_IWUSR) + + +def export_qdrant_memories(args: argparse.Namespace, anon_ids: Dict[str, Any]) -> Tuple[Path, int]: + base_url = qdrant_url(args) + api_key = qdrant_api_key(args) + collection_name = qdrant_collection(args) + output_path = Path(args.output).expanduser() if args.output else default_export_path() + page_size = args.qdrant_page_size + if page_size < 1: + raise MigrationError("--qdrant-page-size must be greater than 0.") + + scroll_filter = qdrant_filter(args) + preflight_qdrant_collection(base_url, api_key, collection_name) + points = scroll_qdrant_points(base_url, api_key, collection_name, scroll_filter, page_size) + records = [normalize_qdrant_point(point) for point in points] + + artifact = { + "version": 1, + "kind": "mem0_oss_qdrant_export", + "exported_at": utc_now(), + "source": { + "sdk": "python", + "vector_store": "qdrant", + "storage": "hosted", + "collection": collection_name, + "url": base_url, + "filters": { + "user_id": args.user_id, + "agent_id": args.agent_id, + "run_id": args.run_id, + "all": args.all, + }, + }, + "local_anonymous_id": primary_anon_id(anon_ids), + "anonymous_ids": { + "oss": anon_ids.get("oss"), + "cli": anon_ids.get("cli"), + }, + "record_count": len(records), + "records": records, + } + write_export_file(output_path, artifact) + return output_path, len(records) + + +def platform_request_json( + method: str, + url: str, + *, + api_key: str, + payload: Optional[Dict[str, Any]] = None, +) -> Dict[str, Any]: + try: + return request_json(method, url, payload=payload, headers=platform_headers(api_key), timeout=30.0) + except HTTPStatusError as exc: + if exc.status_code in (401, 403): + raise AuthError("Platform authentication failed. Check your MEM0_API_KEY or rerun email login.") from exc + raise MigrationError(f"Platform request failed: {exc.detail}") from exc + + +def import_input_path(args: argparse.Namespace) -> Path: + if not args.input: + raise MigrationError("--input is required for --import-only.") + return Path(args.input).expanduser() + + +def read_import_records(input_path: Path) -> Tuple[Dict[str, Any], List[Dict[str, Any]]]: + try: + parsed = json.loads(input_path.read_text(encoding="utf-8")) + except FileNotFoundError as exc: + raise MigrationError(f"Import file not found: {input_path}") from exc + except json.JSONDecodeError as exc: + raise MigrationError(f"Import file is not valid JSON: {input_path}") from exc + + if isinstance(parsed, list): + records = parsed + artifact: Dict[str, Any] = {} + elif isinstance(parsed, dict): + artifact = parsed + raw_records = parsed.get("records") + if not isinstance(raw_records, list): + raise MigrationError("Import artifact must contain a records array.") + records = raw_records + else: + raise MigrationError("Import file must be a JSON object or array.") + + return artifact, [record for record in records if isinstance(record, dict)] + + +def migration_import_key(source: Dict[str, Any], record: Dict[str, Any]) -> str: + raw = ":".join( + [ + str(source.get("sdk") or "unknown"), + str(source.get("vector_store") or "unknown"), + str(source.get("collection") or "unknown"), + str(record.get("id") or ""), + ] + ) + return sha256(raw.encode("utf-8")).hexdigest() + + +def parse_timestamp(value: Any) -> Optional[int]: + if not isinstance(value, str) or not value: + return None + try: + return int(datetime.fromisoformat(value.replace("Z", "+00:00")).timestamp()) + except ValueError: + return None + + +def build_import_metadata(source: Dict[str, Any], record: Dict[str, Any]) -> Dict[str, Any]: + existing_metadata = record.get("metadata") if isinstance(record.get("metadata"), dict) else {} + metadata = dict(existing_metadata) + metadata.update( + { + "mem0_migration_source": record.get("source") or "python_oss_qdrant", + "mem0_migration_collection": source.get("collection"), + "mem0_migration_local_id": record.get("id"), + "mem0_migration_local_hash": record.get("hash"), + "mem0_migration_import_key": migration_import_key(source, record), + } + ) + return {key: value for key, value in metadata.items() if value is not None} + + +def build_import_payload(source: Dict[str, Any], record: Dict[str, Any]) -> Dict[str, Any]: + memory = record.get("memory") + if not isinstance(memory, str) or not memory.strip(): + raise MigrationError("record is missing memory text") + + payload: Dict[str, Any] = { + "messages": [{"role": "user", "content": memory}], + "metadata": build_import_metadata(source, record), + "infer": False, + "source": "migration", + } + + for key in ("user_id", "agent_id", "run_id"): + value = record.get(key) + if value: + payload[key] = value + + timestamp = parse_timestamp(record.get("created_at")) + if timestamp is not None: + payload["timestamp"] = timestamp + + if not any(payload.get(key) for key in ("user_id", "agent_id", "run_id")): + raise MigrationError("record must have at least one of user_id, agent_id, or run_id") + + return payload + + +def payload_scope(payload: Dict[str, Any]) -> Tuple[str, str]: + for key in ("user_id", "agent_id", "run_id"): + value = payload.get(key) + if value: + return key, str(value) + raise MigrationError("import payload must have at least one of user_id, agent_id, or run_id") + + +def platform_get_all_scope(base_url: str, api_key: str, scope_field: str, scope_value: str) -> List[Dict[str, Any]]: + memories: List[Dict[str, Any]] = [] + page = 1 + while True: + response = platform_request_json( + "POST", + f"{base_url}/v3/memories/?page={page}&page_size={DEFAULT_PLATFORM_PAGE_SIZE}", + api_key=api_key, + payload={"filters": {scope_field: scope_value}, "source": "migration"}, + ) + page_results = response.get("results") + if not isinstance(page_results, list): + raise MigrationError("Platform get_all response did not include a results list.") + memories.extend(memory for memory in page_results if isinstance(memory, dict)) + if not response.get("next"): + return memories + page += 1 + + +def existing_migration_memories( + base_url: str, + api_key: str, + payloads: List[Dict[str, Any]], +) -> Dict[str, Dict[str, Any]]: + scopes: List[Tuple[str, str]] = [] + seen_scopes: Set[Tuple[str, str]] = set() + for payload in payloads: + scope = payload_scope(payload) + if scope not in seen_scopes: + seen_scopes.add(scope) + scopes.append(scope) + + existing: Dict[str, Dict[str, Any]] = {} + for scope_field, scope_value in scopes: + print_info(f"Checking existing Platform memories for {scope_field}={scope_value!r}...") + for memory in platform_get_all_scope(base_url, api_key, scope_field, scope_value): + metadata = memory.get("metadata") + if not isinstance(metadata, dict): + continue + key = metadata.get("mem0_migration_import_key") + if isinstance(key, str) and key: + existing[key] = memory + return existing + + +def review_file_path(input_path: Path) -> Path: + timestamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") + return input_path.with_name(f"{input_path.stem}-import-review-{timestamp}.json") + + +def write_import_review_file(input_path: Path, review_records: List[Dict[str, Any]], summary: Dict[str, Any]) -> Optional[Path]: + if not review_records: + return None + path = review_file_path(input_path) + payload = { + "version": 1, + "kind": "mem0_platform_import_review", + "created_at": utc_now(), + "summary": summary, + "records": review_records, + } + path.write_text(json.dumps(payload, indent=2) + "\n", encoding="utf-8") + os.chmod(path, stat.S_IRUSR | stat.S_IWUSR) + return path + + +def import_platform_memories(input_path: Path, api_key: str, base_url: str) -> Dict[str, Any]: + artifact, records = read_import_records(input_path) + source = artifact.get("source") if isinstance(artifact.get("source"), dict) else {} + + payloads: List[Tuple[Dict[str, Any], Dict[str, Any]]] = [] + review_records: List[Dict[str, Any]] = [] + invalid = 0 + for record in records: + try: + payloads.append((record, build_import_payload(source, record))) + except MigrationError as exc: + invalid += 1 + review_records.append({"status": "invalid", "error": str(exc), "record": record}) + + existing = existing_migration_memories(base_url, api_key, [payload for _record, payload in payloads]) if payloads else {} + + imported = 0 + skipped_existing = 0 + changed_existing = 0 + failed = 0 + statuses: Counter[str] = Counter() + + for record, payload in payloads: + key = payload["metadata"]["mem0_migration_import_key"] + existing_memory = existing.get(key) + if existing_memory: + existing_metadata = existing_memory.get("metadata") if isinstance(existing_memory.get("metadata"), dict) else {} + existing_hash = existing_metadata.get("mem0_migration_local_hash") + local_hash = payload["metadata"].get("mem0_migration_local_hash") + if existing_hash == local_hash: + skipped_existing += 1 + else: + changed_existing += 1 + review_records.append( + { + "status": "changed_existing", + "platform_memory_id": existing_memory.get("id"), + "existing_hash": existing_hash, + "local_hash": local_hash, + "record": record, + } + ) + continue + + try: + response = platform_request_json("POST", f"{base_url}/v3/memories/add/", api_key=api_key, payload=payload) + imported += 1 + statuses[str(response.get("status", "unknown"))] += 1 + except MigrationError as exc: + failed += 1 + review_records.append({"status": "failed", "error": str(exc), "record": record}) + + summary = { + "input_records": len(records), + "valid_records": len(payloads), + "invalid": invalid, + "imported": imported, + "skipped_existing": skipped_existing, + "changed_existing": changed_existing, + "failed": failed, + "response_statuses": dict(statuses), + } + review_path = write_import_review_file(input_path, review_records, summary) + summary["review_path"] = str(review_path) if review_path else None + return summary + + +def print_import_summary(summary: Dict[str, Any]) -> None: + print_info(f"Imported: {summary['imported']}") + print_info(f"Skipped existing identical: {summary['skipped_existing']}") + print_info(f"Changed existing: {summary['changed_existing']}") + print_info(f"Invalid: {summary['invalid']}") + print_info(f"Failed: {summary['failed']}") + if summary.get("review_path"): + print_info(f"Review file: {summary['review_path']}") + + +def parse_args(argv: List[str]) -> argparse.Namespace: + parser = argparse.ArgumentParser( + prog="oss-to-platform-migrate.sh", + description="Migrate Python OSS hosted-Qdrant memories to a Mem0 Platform account.", + ) + parser.add_argument("--auth-only", action="store_true", help="Run only auth/account resolution.") + parser.add_argument("--export-only", action="store_true", help="Run only Python OSS hosted-Qdrant JSON export.") + parser.add_argument("--import-only", action="store_true", help="Run only Platform import from a migration JSON file.") + parser.add_argument("--email", help="Email address for email-code authentication.") + parser.add_argument("--code", help="Verification code for non-interactive email-code authentication.") + parser.add_argument("--api-key", help="Platform API key to use for this run only.") + parser.add_argument("--base-url", help=f"Mem0 Platform API base URL. Defaults to {DEFAULT_BASE_URL}.") + parser.add_argument("--yes", action="store_true", help="Accept an existing valid Platform session without prompting.") + parser.add_argument("--qdrant-url", help="Hosted Qdrant URL for Python OSS memory export.") + parser.add_argument("--qdrant-api-key", help="Hosted Qdrant API key for this run only.") + parser.add_argument("--qdrant-collection", help=f"Qdrant collection name. Defaults to {DEFAULT_QDRANT_COLLECTION}.") + parser.add_argument("--qdrant-page-size", type=int, default=DEFAULT_EXPORT_PAGE_SIZE, help="Qdrant scroll page size.") + parser.add_argument("--user-id", help="Export memories matching this user_id.") + parser.add_argument("--agent-id", help="Export memories matching this agent_id.") + parser.add_argument("--run-id", help="Export memories matching this run_id.") + parser.add_argument("--all", action="store_true", help="Export the full Qdrant collection without a scope filter.") + parser.add_argument("--output", help="Path for the migration JSON export file.") + parser.add_argument("--input", help="Path to a migration JSON file to import into Platform.") + return parser.parse_args(argv) + + +def main(argv: List[str]) -> int: + args = parse_args(argv) + exclusive_modes = [args.auth_only, args.export_only, args.import_only] + if sum(1 for enabled in exclusive_modes if enabled) > 1: + print_error("Use only one of --auth-only, --export-only, or --import-only.") + return 1 + + path = config_path() + config = load_config(path) + oss_anon_id = ensure_oss_anon_id(path, config) + anon_ids = read_anon_ids(config, oss_anon_id) + base_url = api_base_url(args, config) + + print_info("Mem0 OSS to Platform migration") + if args.export_only: + print_info("Export phase") + elif args.import_only: + print_info("Import phase") + else: + print_info("Auth phase") + print_info("Supported environments: macOS, Linux, and Windows via WSL/Git Bash with bash and python3.") + if args.auth_only: + print_info("No memories are imported during this auth step.") + elif args.export_only: + print_info("No memories are imported to Platform during this step.") + else: + print_info("Memories will be imported to your authenticated Mem0 Platform account.") + print_info("") + + capture_migration_event("oss.migrate.started", anon_ids, additional={"base_url": base_url}) + + try: + email = None + api_key = None + output_path = None + export_count = 0 + import_summary: Optional[Dict[str, Any]] = None + + if args.import_only: + input_path = import_input_path(args) + api_key, key_source = configured_api_key(args, config) + if not api_key: + raise MigrationError("--import-only requires --api-key, MEM0_API_KEY, or a stored Platform API key.") + try: + email = validate_api_key(api_key, base_url) + except MigrationError: + if key_source in ("flag", "env"): + raise + email = None + if email: + capture_migration_event( + "oss.migrate.authenticated", + anon_ids, + email=email, + additional={"auth_method": f"{key_source}_api_key", "base_url": base_url}, + ) + stitch_identities(path, config, anon_ids, email) + print_info("Phase 1/1: Import memories into Mem0 Platform") + import_summary = import_platform_memories(input_path, api_key, base_url) + elif not args.export_only: + phase_label = "Phase 1/1" if args.auth_only else "Phase 1/3" + print_info(f"{phase_label}: Authenticate with Mem0 Platform") + api_key, email, auth_method = resolve_auth(args, config, base_url) + capture_migration_event( + "oss.migrate.authenticated", + anon_ids, + email=email, + additional={"auth_method": auth_method, "base_url": base_url}, + ) + stitch_identities(path, config, anon_ids, email) + print_info(f"Authenticated as {email}") + print_info("") + + if not args.auth_only: + if args.export_only: + print_info("Phase 1/1: Export Python OSS memories from hosted Qdrant") + output_path, export_count = export_qdrant_memories(args, anon_ids) + elif not args.import_only: + print_info("Phase 2/3: Export Python OSS memories from hosted Qdrant") + output_path, export_count = export_qdrant_memories(args, anon_ids) + print_info(f"Exported {export_count} memories to {output_path}") + print_info("") + print_info("Phase 3/3: Import memories into Mem0 Platform") + import_summary = import_platform_memories(output_path, api_key or "", base_url) + + if import_summary is not None: + capture_migration_event( + "oss.migrate.completed", + anon_ids, + email=email, + additional={ + "base_url": base_url, + "phase": "import", + "exported": export_count, + "imported": import_summary["imported"], + "skipped_existing": import_summary["skipped_existing"], + "changed_existing": import_summary["changed_existing"], + "invalid": import_summary["invalid"], + "failed": import_summary["failed"], + }, + ) + except MigrationError as exc: + capture_migration_event( + "oss.migrate.failed", + anon_ids, + email=email, + additional={"error": str(exc), "base_url": base_url}, + ) + print_error(str(exc)) + return 1 + + if args.auth_only: + print_info("Auth-only mode complete. Export and import were not run.") + elif args.export_only: + print_info(f"Exported {export_count} memories to {output_path}") + print_info("Export-only mode complete. Platform import was not run.") + elif args.import_only: + print_import_summary(import_summary or {}) + print_info("Import-only mode complete. Export was not run.") + else: + print_info("") + print_info(f"Export file: {output_path}") + print_info(f"Exported: {export_count}") + print_import_summary(import_summary or {}) + print_info("Migration complete.") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main(sys.argv[1:])) +PY diff --git a/skills/README.md b/skills/README.md new file mode 100644 index 000000000..c14b5cab2 --- /dev/null +++ b/skills/README.md @@ -0,0 +1,49 @@ +# Mem0 Skills for AI Coding Assistants + +Mem0 ships structured skill definitions for Claude Code, Codex, Cursor, OpenCode, OpenClaw, and any assistant that supports the [skills standard](https://github.com/anthropic-experimental/skills). Skills teach the assistant how to work with Mem0 — either by loading SDK knowledge into context, or by executing an end-to-end workflow on demand. + +## Two Categories + +### Reference skills — always on + +Installed once, loaded into context so the assistant writes correct Mem0 code. Use these for day-to-day development. + +| Skill | Surface | Install | +|-------|---------|---------| +| [`mem0`](./mem0/) | Python + TypeScript SDKs (Platform + OSS), framework integrations | `npx skills add https://github.com/mem0ai/mem0 --skill mem0` | +| [`mem0-cli`](./mem0-cli/) | Terminal workflows (`mem0` CLI, both Node and Python) | `npx skills add https://github.com/mem0ai/mem0 --skill mem0-cli` | +| [`mem0-vercel-ai-sdk`](./mem0-vercel-ai-sdk/) | `@mem0/vercel-ai-provider` and `createMem0` | `npx skills add https://github.com/mem0ai/mem0 --skill mem0-vercel-ai-sdk` | + +### Pipeline skills — run on demand + +Invoked as a slash command to execute a specific end-to-end workflow. These do real work: they create branches, write tests, run code. + +| Skill | Trigger | Install | +|-------|---------|---------| +| [`mem0-integrate`](./mem0-integrate/) | `/mem0-integrate` — wire Mem0 into an existing repo via TDD | `npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate` | +| [`mem0-test-integration`](./mem0-test-integration/) | `/mem0-test-integration` — verify what `/mem0-integrate` produced | `npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration` | + +The two pipeline skills are designed to run in sequence on the same workspace: + +``` +/mem0-integrate → mem0-integrate/ branch + .mem0-integration/ artifacts +/mem0-test-integration → scorecard (compile + runtime verification, real API smoke test) +``` + +## Choosing a Skill + +- **Writing Mem0 code in a new or existing project?** → `mem0` +- **Using the terminal CLI?** → `mem0-cli` +- **Building with `@ai-sdk/*`?** → `mem0-vercel-ai-sdk` +- **Want the assistant to wire Mem0 into an existing repo for you?** → `mem0-integrate`, then `mem0-test-integration` + +## Links + +- [Vibecoding with Mem0](https://docs.mem0.ai/vibecoding) — canonical landing page +- [Claude Code integration](https://docs.mem0.ai/integrations/claude-code) +- [Mem0 Platform Dashboard](https://app.mem0.ai) +- [Mem0 Documentation](https://docs.mem0.ai) + +## License + +Apache-2.0 diff --git a/skills/mem0-integrate/LICENSE b/skills/mem0-integrate/LICENSE new file mode 100644 index 000000000..78c99ae28 --- /dev/null +++ b/skills/mem0-integrate/LICENSE @@ -0,0 +1,189 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but not + limited to compiled object code, generated documentation, and + conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work. + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to the Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by the Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding any notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + Copyright 2024 Mem0.ai + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/skills/mem0-integrate/README.md b/skills/mem0-integrate/README.md new file mode 100644 index 000000000..3c8e47478 --- /dev/null +++ b/skills/mem0-integrate/README.md @@ -0,0 +1,89 @@ +# mem0-integrate — Pipeline Skill + +Wire [Mem0](https://mem0.ai) into an existing repository end-to-end, using a goal-driven, test-first pipeline. + +> **This is a pipeline skill, not a reference skill.** Invoke it as `/mem0-integrate` when you want your assistant to do the work of integrating Mem0 into a target repo. For day-to-day SDK coding help, install [`mem0`](../mem0/SKILL.md) instead. +> +> **Part of the Mem0 Skill Graph:** +> - Reference: [mem0](../mem0/SKILL.md) · [mem0-cli](../mem0-cli/SKILL.md) · [mem0-vercel-ai-sdk](../mem0-vercel-ai-sdk/SKILL.md) +> - Pipeline: **mem0-integrate** (this skill) → [mem0-test-integration](../mem0-test-integration/SKILL.md) + +## What This Skill Does + +When invoked, your assistant will: + +- **Detect** the target repo's language and stack automatically +- **Ask** whether to integrate with Mem0 Platform (managed) or Mem0 Open Source (self-hosted) +- **Write failing tests first** — no implementation until tests exist +- **Keep the integration additive and feature-flagged** — existing behavior stays byte-for-byte identical when the flag is unset +- **Produce a local feature branch** (`mem0-integrate/...`) and a `.mem0-integration/` directory of artifacts (`goal.md`, `plan.md`, `product.json`) consumed by the companion verification skill + +## When to Use + +Trigger phrases: + +- "Integrate Mem0 into this repo" +- "Add Mem0 to my project" +- "Wire Mem0 into ``" +- "How do I add memory to an existing project?" + +Do **not** use this skill for general SDK usage (install [`mem0`](../mem0/SKILL.md)), terminal workflows (install [`mem0-cli`](../mem0-cli/SKILL.md)), or Vercel AI SDK integration (install [`mem0-vercel-ai-sdk`](../mem0-vercel-ai-sdk/SKILL.md)). + +## Installation + +### CLI (Claude Code, Codex, OpenCode, OpenClaw, or any tool that supports skills) + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate +``` + +For verification on the same branch, also install the companion skill: + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration +``` + +### Claude.ai + +1. Download this `skills/mem0-integrate` folder as a ZIP +2. Go to **Settings > Capabilities > Skills** +3. Click **Upload skill** and select the ZIP + +### Claude API (Skills API) + +```bash +curl -X POST https://api.anthropic.com/v1/skills \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"name": "mem0-integrate", "source": "https://github.com/mem0ai/mem0/tree/main/skills/mem0-integrate"}' +``` + +### Prerequisites + +- A Mem0 Platform API key ([get one](https://app.mem0.ai/dashboard/api-keys)) *or* a working OSS setup (LLM + vector store) +- Python 3.10+ or Node.js 18+ in the target repo +- A clean working tree on the target repo's default branch + +## Workflow + +``` +/mem0-integrate → creates mem0-integrate/ branch, + writes .mem0-integration/ artifacts, + implements against failing tests +/mem0-test-integration → runs the repo's native test suite, + executes a real end-to-end smoke flow, + produces a scorecard +``` + +The two skills are loosely coupled — they share the same workspace and branch via `.mem0-integration/`, but the verifier never modifies source. + +## Links + +- [Mem0 Platform Dashboard](https://app.mem0.ai) +- [Mem0 Documentation](https://docs.mem0.ai) +- [Mem0 GitHub](https://github.com/mem0ai/mem0) +- [Platform vs OSS comparison](https://docs.mem0.ai/platform/platform-vs-oss) + +## License + +Apache-2.0 diff --git a/skills/mem0-integrate/SKILL.md b/skills/mem0-integrate/SKILL.md new file mode 100644 index 000000000..8230a63b3 --- /dev/null +++ b/skills/mem0-integrate/SKILL.md @@ -0,0 +1,620 @@ +--- +name: mem0-integrate +description: > + Integrate Mem0 into an existing repository using a goal-driven, TDD pipeline. + Detects the repo's language automatically and asks the user to pick between + Mem0 Platform (managed) and Mem0 Open Source (self-hosted). Writes failing + tests before any implementation. Produces a local feature branch plus + `.mem0-integration/` artifacts consumed by the paired verification skill. + TRIGGER when: user says "integrate mem0", "add mem0 to this repo", "wire + mem0 into ", or asks how to add memory to an existing project. + DO NOT TRIGGER when: the user wants general SDK usage (use skill:mem0), + CLI usage (use skill:mem0-cli), or Vercel AI SDK (use skill:mem0-vercel-ai-sdk). + After success, invoke skill:mem0-test-integration to verify in the same + workspace (loose coupling). +license: Apache-2.0 +metadata: + author: mem0ai + version: "0.1.0" + category: ai-memory + tags: "memory, integration, tdd, platform, oss" + mem0_tested_versions: "mem0ai (PyPI) >=2.0.0,<3.0.0; mem0ai (npm) >=3.0.0,<4.0.0" +--- + +# mem0-integrate + +Wire Mem0 into an existing repo with a goal-driven, test-first pipeline. +Pairs with `mem0-test-integration` for verification. + +## Canonical sources (fetch before deciding anything) + +The skill MUST `WebFetch` these URLs before step 3 and cite them in +`plan.md`. They are the ground truth — do not rely on ambient knowledge +of the Mem0 API. + +### Agent-ready docs +- Scope-tagged docs index: https://docs.mem0.ai/llms.txt +- Full docs (single file, deep dives): https://docs.mem0.ai/llms-full.txt +- OpenAPI spec (Platform REST, machine-readable): https://docs.mem0.ai/openapi.json +- Hosted MCP server: https://mcp.mem0.ai (requires Platform API key) +- Integrations index: https://docs.mem0.ai/integrations + +### Published Mem0 skills — delegate; do not reimplement +Prefer these over writing your own call-site patterns. Each is a +standalone `SKILL.md` with triggers, examples, and version-pinned code. + +- SDK (Python + TS, Platform + OSS): https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0/SKILL.md +- CLI: https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0-cli/SKILL.md +- Vercel AI SDK: https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0-vercel-ai-sdk/SKILL.md +- Editor/MCP plugin glue (9 MCP tools): https://github.com/mem0ai/mem0/tree/main/mem0-plugin + +### SDK source (read when docs are ambiguous) +Public repo. Cross-check against the `mem0_tested_versions` range in this +skill's frontmatter if the `main` branch has moved past a major. + +- Repo root: https://github.com/mem0ai/mem0 +- Python SDK: https://github.com/mem0ai/mem0/tree/main/mem0 +- TypeScript SDK: https://github.com/mem0ai/mem0/tree/main/mem0-ts + +### Quickstarts (for bootstrapping unfamiliar stacks) +- Platform: https://docs.mem0.ai/platform/quickstart +- OSS Python: https://docs.mem0.ai/open-source/python-quickstart +- OSS Node: https://docs.mem0.ai/open-source/node-quickstart +- Platform vs OSS comparison: https://docs.mem0.ai/platform/platform-vs-oss + +## Integration principles (non-negotiable) + +The true goal of this skill is to produce a **PR the maintainers can accept +without argument**. That rules out anything invasive. + +1. **Additive, not replacing.** If the target repo already has a memory + system, a session store, a user-context layer, or anything named + `Memory` / `memory_*`, Mem0 sits **alongside** it, not in place of it. + The existing system keeps working unchanged. +2. **Opt-in by default.** Gate all new Mem0 code behind a feature flag + (env var like `MEM0_ENABLED=1`, a config key, or a strategy selector). + With the flag unset, behavior is the repo's original behavior, + byte-for-byte. +3. **No breakage.** No removed exports, no renamed public functions, + no changed method signatures, no modified existing tests, no changed + behavior of existing tests. All pre-existing tests must pass unchanged + both with the flag set and unset. +4. **Minimal dependency surface.** Add `mem0ai` (plus any deps the + delegated skill requires) and nothing else. No new vector stores, no + graph databases, no provider SDKs the repo does not already use. +5. **Separable commits.** Code, tests, and config/docs land in separate + commits so reviewers can cherry-pick. +6. **The null hypothesis wins.** If no additive, gated fit exists after + step 6 (plan), exit with code 1 and a rationale. A bad PR is worse + than no PR. +7. **Backend only.** Mem0 integration lives in server-side code. API keys, + memory scope, and user-identity resolution are not safe client-side. + If the repo has both backend and frontend, the call sites live in + backend files. Frontend-only repos are rejected at preconditions. + +Enforced at four gates: **preconditions** (reject frontend-only repos +and repos where additive fit is impossible), **step 2 comprehension** +(confirm a backend exists and name candidate surfaces), **step 6 plan +review** (reject plans that mutate existing exports or name client-side +call sites), and **step 10 self-healing loop** (refuse to "fix" principle +violations — surface them instead). + +## Skill delegation rules + +Before writing any code, check whether a published skill already covers +the target stack. If yes, delegate — copy its call-site pattern into +`plan.md` and into the tests; do not paraphrase. + +| Detected in target repo | Delegate to | Why | +|---|---|---| +| `@ai-sdk/*` + `ai` in `package.json` | `skills/mem0-vercel-ai-sdk` | Integration is via `createMem0` provider wrapper, not raw `MemoryClient`. | +| CLI-only repo (Typer, Commander, Click, Cobra) with no LLM call sites | `skills/mem0-cli` | Call sites are command handlers, not model wrappers. Consider whether mem0 actually fits first. | +| Target is an MCP client / editor config (Claude Code, Cursor, Codex settings) | `mem0-plugin` | Wire via MCP server URL + hooks; no SDK code usually needed. | +| Any other Python or TS repo with an LLM call site | `skills/mem0` | Default SDK integration path. | + +Record the delegated skill's raw URL in `plan.md` under a +**"Delegated skill:"** field. The test writer in step 7 and the +implementation subagent in step 8 both read this field. + +## Preconditions + +Refuse to start unless ALL of the following are true: + +- Current working directory is inside a git repository with a clean index + (no uncommitted changes). Protects the user's work — every edit lands on + a feature branch, not on top of in-progress changes. +- Repo has a detectable language (`package.json` / `pyproject.toml` / + `requirements.txt`). No language → exit cleanly with a written rationale. +- Repo has a **backend**. Detected by: a `backend/` or `server/` or `api/` + directory; a Python package with FastAPI/Flask/Django/Starlette; a Node + package with Express/Fastify/Koa/NestJS/Next-API-routes; an agent-loop + framework (LangGraph, LangChain, LlamaIndex, Agno). Frontend-only repos + (pure React/Vue/Svelte SPAs, static sites, mobile-only) → exit with + code 1 and a rationale. Mem0 is not installed client-side. +- The user has already decided Mem0 fits this repo. This skill does NOT + survey the codebase to justify fit — bring a concrete goal. (Step 2 + *does* read the repo to understand what it does and locate backend + integration surfaces; that is mechanics, not fit-justification.) + +Exit with a written rationale if any precondition fails. Do not try to +"make it work anyway." + +## Pipeline + +### 1. Language detection + +| Signal | Track | +|---|---| +| `package.json` + TypeScript config | Node / TypeScript | +| `package.json` (no TS config) | Node / JavaScript | +| `pyproject.toml` or `requirements.txt` | Python | + +Monorepo with both → ask which subdirectory to operate in, then recurse. + +### 2. Repo comprehension — what does this repo do, and where is the backend? + +Before any decision (product, goal, plan), understand the repo enough +to locate *where in the backend* the integration belongs. This is not +fit-surveying — the user already decided Mem0 fits. This is mechanics: +you cannot write a plan without knowing what files matter. + +Read, in order, with a token budget — do not scan the whole tree: + +1. `README.md` (root) + first-page of any `README_*.md` variants. +2. `CONTRIBUTING.md` / `AGENTS.md` / `CLAUDE.md` at root if present — + these often spell out architecture and entry points. +3. `package.json` / `pyproject.toml` scripts + entry points. +4. The layout of the top two directory levels (not recursive). +5. Key config files: `docker-compose.yml`, `Dockerfile`, `Makefile`, + `langgraph.json`, `next.config.*`, `nuxt.config.*`. + +Produce `.mem0-integration/repo-summary.md`: + + # Repo comprehension + + **What this repo does:** + + **Architecture at a glance:** + - Backend: + - Frontend: + - Agent loop / orchestration: + - Existing memory/session/state systems: + + **Candidate backend integration surfaces** (ranked, best first): + 1. `:` — — + 2. ... + 3. ... + + **Not a fit here:** + + **Sources read:** + +Show the user the rendered summary and ask: *"Is this understanding +correct? Which of the candidate surfaces (1, 2, 3 ...) should step 3 +forward target?"* + +Gate rules: + +- If no backend surface is found → exit code 1. The preconditions + should already have caught frontend-only repos; reaching this point + means a more subtle miss (e.g., the "backend" is actually just a + static build). Do not force a fit. +- If every candidate surface would require replacing an existing + memory/session system → exit code 1 with the "additive principle" + rationale. The user can manually point at a non-conflicting location + and re-run. +- User corrections update `repo-summary.md` and re-confirm. Max 3 + rounds; beyond that, exit code 1. + +The user's chosen surface index is baked into `product.json` as +`preferred_site` and referenced by steps 5 and 6. + +### 3. Product selection — Platform vs OSS (ask with a recommendation) + +Read the `## Identify the User's Setup` block in +`https://docs.mem0.ai/llms.txt` for the Platform-first routing rules, then +apply the heuristics below. Ask, but never blank: + +- Other managed-service SDKs present (`@clerk/*`, `stripe`, `@supabase/*`, + `openai`, `@upstash/*`, `posthog-*`) — 3+ → recommend **Platform**. +- Local-infra signals (`docker-compose.yml` with postgres / redis / qdrant / + neo4j, ollama configs, self-hosted auth) — 2+ → recommend **OSS**. +- No strong signal → default recommendation: **Platform** (lower integration + cost; migration later is supported). + +Example: + +> I see `stripe`, `@clerk/nextjs`, and `@supabase/supabase-js` — managed +> services throughout. I recommend **Mem0 Platform** (4-line integration). +> Override and use open source? + +Bake the choice into the goal doc in step 5. Do not re-decide later. + +### 4. API key check (env-first, then ask) + +| Track | Key | Where to find | +|---|---|---| +| Platform | `MEM0_API_KEY` | https://app.mem0.ai | +| OSS (default LLM) | `OPENAI_API_KEY` | https://platform.openai.com/api-keys | + +If present in env → continue. +If missing → **interactive mode** asks; **CI mode** (`MEM0_INTEGRATE_CI=1`) +exits with code 2 and the name of the missing key. + +Never echo key values into `trace.jsonl`. Persist to `.env` only with +explicit user consent, and append `.env` to `.gitignore` if not already there. + +If the user is on OSS and wants a non-OpenAI LLM, route them to the +`components/llms/*` docs and re-run this step with the chosen provider's key. + +### 5. Goal doc — the hard gate + +Write `.mem0-integration/goal.md` and **require user approval before step 6**. + +Template: + + # Mem0 Integration Goal + + **What gets stored:** + + **When it gets retrieved:** + + **Why:** + + **Product:** Platform | OSS (locked from step 3, do not change) + + **Delegated skill:** . + + **Out of scope:** + +Rules: + +- User must approve explicitly. If they edit the doc, reload and re-confirm. +- `goal.md` is the contract the test suite is written against. Never + rewrite it after step 6 starts. +- Max 3 rejection rounds. On the 4th, exit with code 3 and the rejection + notes — the integration is not well-specified enough to proceed. + +### 6. Integration plan — how and where (hard gate) + +Given `goal.md` is "what and why," this step produces "where and how" and +gets explicit user sign-off before any code is written. + +The skill does a **scoped** read of the repo (no wide survey): + +- Grep for the LLM call sites that match the goal (e.g., `openai.chat.`, + `anthropic.messages.`, `model.generateContent`, `ChatOpenAI`, `createLLM`). +- Grep for the user-identity source (`req.user`, `session.user`, `auth()`, + `ctx.userId`, cookies). +- Check `package.json` / `pyproject.toml` / `requirements.txt` for + conflicts (e.g., existing `mem0ai` at a different version). + +Then write `.mem0-integration/plan.md`: + + # Mem0 Integration Plan + + **Write pattern:** )."> + + **Read pattern:** , limit=5) + and inject results as a system message."> + + **User identifier source:** + + **Session scoping:** + - user_id: + - agent_id: + - run_id: + + **Write call site:** `` — inside `` + **Read call site:** `` — inside `` + + **Dependencies to add:** + - `@` + + **Preserved behavior:** + + **Coexistence:** + + **Feature flag:** + + **Sources consulted:** + + **E2E recipe:** + + start: + ready_probe: status= / + log="" / + sleep=> + compose_services: + write_call: + write_async_wait_ms: + read_call: + read_assert: = + that MUST appear in read_call's output for + the E2E to pass. Derived from goal.md's + "What gets stored."> + + **Rejected alternatives:** + +Rules: + +- Show the user the proposed call sites with 10 lines of context around + each before asking for approval. +- If the skill can't find a plausible call site for either write or read, + it exits with code 5 and asks the user to name the file(s) manually + (this is the "no fit here" signal — don't guess). +- Max 3 rejection rounds on the plan. On the 4th, exit code 5 with the + last plan and the user's notes. +- If the user edits `plan.md` by hand, reload and re-confirm. + +`plan.md` (not `goal.md`) is the contract the subagent implements against +in step 8. + +### 7. Tests first (TDD) + +Main agent writes failing tests against `goal.md` in the repo's native +test framework: + +| Track | Default framework | +|---|---| +| Python | `pytest` | +| TypeScript | `vitest` if detected, else `jest` | +| JavaScript | same | + +Test assertion shapes must match the **canonical signatures**: + +- Platform method signatures: `https://docs.mem0.ai/openapi.json` + (request body schemas for `/v1/memories/` and `/v1/memories/search/`). +- OSS method signatures: the delegated skill named in `plan.md` + (fetched from its raw URL) or `skills/mem0/SKILL.md` as the default. +- Do not hand-roll request shapes. If the delegated skill has an + example block, lift it verbatim. + +Minimum two test files (paths taken from `plan.md` call sites): + +- `test_mem0_write.` — asserts `add()` is called at the Write call + site with the right payload shape (Platform messages-array vs OSS string) + and the right `user_id` source. +- `test_mem0_read.` — asserts `search()` runs before the Read call + site and the result is wired into the LLM prompt / response path. + +Tests MUST be importable with `MEM0_API_KEY` unset. This is the design +pressure that forces step 8's lazy `MemoryClient()` / `Memory()` +construction — eager module-level init hits the API on import and +breaks pre-existing test collection when the key is missing. + +Run the tests. They **must fail**. If they pass before any implementation, +the tests are wrong — rewrite them. + +### 8. Implementation (subagent, fresh context) + +Spawn a subagent with: + +- **Inputs**: the repo, `goal.md`, `plan.md`, the two test files, and + direct URLs to: the delegated skill (from `plan.md`), the SDK source + (pinned per `mem0_tested_versions`), `https://docs.mem0.ai/llms.txt`, + and `https://docs.mem0.ai/openapi.json`. +- **No access** to main agent's reasoning trace or scratchpad. +- **System prompt** (verbatim): + + You are implementing a Mem0 integration for an existing repo. + + Read these first: + - plan.md (the mechanical contract) + - goal.md (the intent — do not change it) + - the test files (do not change them either) + - + - https://docs.mem0.ai/llms.txt + - https://docs.mem0.ai/openapi.json (Platform only) + + Constraints — all required, all enforced at review: + + 1. Touch only the files named in plan.md's call sites, or add + strictly new files. + 2. Do not remove or rename any existing symbol. Do not change + any public signature. + 3. Do not modify any existing test. + 4. Gate every line of new Mem0 code behind the feature flag from + plan.md. With the flag in its default state, the repo must + behave exactly like `main` — byte-for-byte, including stdout + and return values. + 5. Use only the SDK surface. No new dependencies + beyond those listed under plan.md's "Dependencies to add." + 6. Preserve everything listed under plan.md's "Preserved behavior" + and "Coexistence." + 7. Lazy client construction. `MemoryClient()` validates the API + key in `__init__` (it makes a network call). Never instantiate + it at module-import time — construct on first use inside the + request / handler path. The same rule applies to OSS `Memory()`, + which can eagerly initialize embedding and LLM providers. Use + a function-local singleton (`functools.lru_cache`, a module-level + `_client = None` + getter, or DI scope) — never a top-level + global. Eager init breaks the pre-existing test suite at + collection time whenever the key is missing or invalid, which + is a non-invasiveness violation. + + Implement the plan to make the new tests pass while all + pre-existing tests continue to pass unchanged. + +Subagent returns a diff. Main agent reviews against `plan.md` (the +mechanical contract) and `goal.md` (the intent): + +- Approved → apply the diff, commit. +- Rejected → return with specific, actionable feedback (not "try again"). +- Max 3 review loops. Beyond that → exit code 4 with the last diff and + reviewer feedback. + +### 9. Commit + handoff + +Create branch `mem0-integrate/` and commit in +**separate commits** so reviewers can cherry-pick: + +1. `mem0: add gated dependency` — just the `pyproject.toml` / `package.json` + change. +2. `mem0: add integration module` — the new file(s). +3. `mem0: wire into ` — the call-site edit(s), still gated. +4. `mem0: add tests` — the new test files. + +If `--no-heal` is set → print `Run /mem0-test-integration to verify.` +and exit. Otherwise proceed to step 10. + +### 10. Self-healing loop (default ON; disable with `--no-heal`) + +Run `/mem0-test-integration --ci` in a subprocess. If `scorecard.json` +reports `overall: pass` → done, exit 0. + +Otherwise loop: + +1. **Categorize the failing check** from `scorecard.json`. Route per + category: + - `install` / `static_checks` → dependency or import fix. + - `unit_tests` → wiring or assertion fix. + - `smoke_test` → API key or SDK call-shape fix. + - `e2e_test` → recipe, flag-wiring, or integration-point fix. + - **Pre-existing test failure (test skill exit code 7, + `non_invasive: false` in scorecard) → STOP.** This is a + non-invasiveness violation. Do NOT attempt to "fix" it (that + breaks principle 3). Exit code 6 with rationale. + +2. **Spawn a remediation subagent**, fresh context. Inputs: + `plan.md`, `goal.md`, `scorecard.md`, `scorecard.json`, the last + committed diff, and the relevant log file for the failing category + (`test-stdout.log` / `smoke-stdout.log` / `e2e-app.log` / + `e2e-calls.log`). + + System prompt (verbatim): + + You are fixing a failing Mem0 integration test. + + Non-negotiable constraints: + - Do not modify test files. + - Do not remove or rename any existing symbol or signature. + - Do not change pre-existing behavior. The feature flag from + plan.md must still default to OFF, and with the flag in its + default state the repo must behave exactly like main. + - Touch only the files named in plan.md's call sites, or add + strictly new files. + - Return the smallest possible diff that fixes the single + failing check listed in scorecard.md. No drive-by cleanup. + +3. **Apply the diff**; commit on the same branch with message + `mem0-heal: attempt `. Do NOT amend earlier commits + (reviewers need the heal trail). + +4. **Re-run `/mem0-test-integration --ci`**. Outcomes: + - `overall: pass` → done, exit 0. + - Same check still failing → increment attempt counter; loop. + - A *different* check now failing → regression. Revert the heal + commit (`git revert HEAD --no-edit`), record the regression in + `.mem0-integration/heal-trace.md`, exit code 6. + +5. **Bounded iterations.** Default 3 attempts per failing category. + Override with `--heal-max N` (hard cap 10). On exhaustion, exit 6 + with the full attempt trace: each diff, each scorecard, final log + tail. + +6. **Post-loop summary** written to `.mem0-integration/heal-trace.md`: + which category failed, how many attempts, each diff's intent, final + status, and — on success — the delta from initial scorecard to final. + +## Artifacts (all under `.mem0-integration/`) + +| File | Purpose | Retention | +|---|---|---| +| `repo-summary.md` | Repo comprehension + candidate backend surfaces (step 2). | Keep across runs. | +| `goal.md` | Approved intent. Never rewritten after step 6. | Keep across runs. | +| `plan.md` | Approved mechanics (where, how, call sites, preserved behavior). | Keep across runs. | +| `trace.jsonl` | Every tool call, decision, and subagent exchange this run. | Overwritten per run. | +| `diff.patch` | The committed integration as a reviewable patch. | Overwritten per run. | +| `heal-trace.md` | Per-attempt record of the self-healing loop (step 10). | Overwritten per run. | +| `product.json` | `{"product": "platform"\|"oss", "language": "...", "mem0_version": "...", "write_site": "file:line", "read_site": "file:line", "feature_flag": "MEM0_ENABLED"}` — consumed by the verification skill. | Overwritten per run. | + +`.mem0-integration/` is added to `.gitignore` on first run. Nothing is +written outside this directory and the repo's source tree. + +## Modes + +| Mode | Trigger | Behavior | +|---|---|---| +| Interactive (default) | TTY present, `MEM0_INTEGRATE_CI` unset | Asks for keys, confirms goal doc, shows recommendations. | +| CI | `MEM0_INTEGRATE_CI=1` | Requires keys in env, requires `--product`, auto-approves goal doc from `goal.md` if present, fails fast otherwise. | + +## Invocation + + /mem0-integrate # interactive, heal ON + /mem0-integrate --no-heal # stop after commit; manual verify + /mem0-integrate --heal-max 5 # cap heal attempts per category (default 3) + /mem0-integrate --product platform # skip the product ask + /mem0-integrate --product oss + /mem0-integrate --ci # non-interactive (for test harness) + +## Exit codes + +| Code | Meaning | +|---|---| +| 0 | Success. Feature branch committed; verification skill ready to run. | +| 1 | Precondition failed (dirty repo, no detectable language, etc.). | +| 2 | Missing env key in CI mode. | +| 3 | Goal doc rejected 3+ times — integration is not well-specified. | +| 4 | Subagent review loop did not converge in 3 rounds. | +| 5 | Integration plan rejected 3+ times, or no plausible additive call site found. | +| 6 | Self-healing loop did not converge, detected a non-invasiveness violation, or a pre-existing test failed. | + +## Explicitly out of scope + +- Surveying the repo for fit points. Humans decide where Mem0 helps before + invoking this skill. +- Replacing any existing memory / session / state system. Always additive + and feature-flagged; see "Integration principles." +- Modifying pre-existing tests, even to "fix" them under self-heal. Tests + that fail after integration with the flag unset are a non-invasiveness + violation, not a bug to patch. +- Deciding Platform vs OSS silently. Always ask with a recommendation. +- Switching branches, pushing, or opening PRs. Commits locally and stops + (or enters the heal loop, still local). +- Data migration between stores. Point user at `migration/oss-to-platform` + docs if they ask. +- Provider selection beyond the default LLM for OSS. If they need a custom + LLM / embedder / vector store, route to `components/*` docs and re-run + step 4 with the new key. diff --git a/skills/mem0-test-integration/LICENSE b/skills/mem0-test-integration/LICENSE new file mode 100644 index 000000000..78c99ae28 --- /dev/null +++ b/skills/mem0-test-integration/LICENSE @@ -0,0 +1,189 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but not + limited to compiled object code, generated documentation, and + conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work. + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to the Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by the Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding any notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + Copyright 2024 Mem0.ai + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/skills/mem0-test-integration/README.md b/skills/mem0-test-integration/README.md new file mode 100644 index 000000000..da905006c --- /dev/null +++ b/skills/mem0-test-integration/README.md @@ -0,0 +1,81 @@ +# mem0-test-integration — Pipeline Skill + +Verify a Mem0 integration produced by [`/mem0-integrate`](../mem0-integrate/SKILL.md). Runs in the same workspace on the same branch — installs dependencies, runs the repo's native test suite, then exercises a real end-to-end smoke flow against the user's API key. + +> **This is a pipeline skill, not a reference skill.** Invoke it as `/mem0-test-integration` after `/mem0-integrate` has produced a branch to verify. It catches compile and runtime bugs by design — logical integration errors (wrong data stored, wrong scoping) are for human review. +> +> **Part of the Mem0 Skill Graph:** +> - Reference: [mem0](../mem0/SKILL.md) · [mem0-cli](../mem0-cli/SKILL.md) · [mem0-vercel-ai-sdk](../mem0-vercel-ai-sdk/SKILL.md) +> - Pipeline: [mem0-integrate](../mem0-integrate/SKILL.md) → **mem0-test-integration** (this skill) + +## What This Skill Does + +When invoked, your assistant will: + +- **Refuse to start** unless the branch has `.mem0-integration/` artifacts, the working tree is clean, and the right API key is in the environment +- **Install** the repo's dependencies using its native tooling (pip, pnpm, npm, hatch, etc.) +- **Run the native test suite** in two passes: flag-unset (must behave like `main`) and flag-set (new tests run) +- **Execute a real end-to-end smoke flow** against Mem0 Platform (`MEM0_API_KEY`) or OSS (`OPENAI_API_KEY`) +- **Produce a scorecard** — `overall: pass | fail`, per-check reasons, and the reproduction command for each failure + +## When to Use + +Trigger phrases: + +- "Verify the integration" +- "Test the Mem0 integration" +- "Run `/mem0-test-integration`" + +Do **not** use this skill to run general project tests (defer to the repo's native test command) or before `/mem0-integrate` has produced a branch on the current workspace. + +## Installation + +### CLI (Claude Code, Codex, OpenCode, OpenClaw, or any tool that supports skills) + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0-test-integration +``` + +Typically installed alongside the companion pipeline skill: + +```bash +npx skills add https://github.com/mem0ai/mem0 --skill mem0-integrate +``` + +### Claude.ai + +1. Download this `skills/mem0-test-integration` folder as a ZIP +2. Go to **Settings > Capabilities > Skills** +3. Click **Upload skill** and select the ZIP + +### Claude API (Skills API) + +```bash +curl -X POST https://api.anthropic.com/v1/skills \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"name": "mem0-test-integration", "source": "https://github.com/mem0ai/mem0/tree/main/skills/mem0-test-integration"}' +``` + +### Preconditions + +The skill refuses to start unless all of the following are true: + +- `.mem0-integration/` directory exists in the repo root +- Current branch starts with `mem0-integrate/` +- Working tree is clean +- The same API key used during `/mem0-integrate` is exported in the environment + +## What This Skill Does *Not* Catch + +By design, this skill only catches compile and runtime bugs. Logical errors — memories stored with the wrong scoping, retrieval returning the wrong user's data, filter mismatches — are the human reviewer's responsibility. + +## Links + +- [Mem0 Documentation](https://docs.mem0.ai) +- [Mem0 GitHub](https://github.com/mem0ai/mem0) +- [API Reference](https://docs.mem0.ai/api-reference) + +## License + +Apache-2.0 diff --git a/skills/mem0-test-integration/SKILL.md b/skills/mem0-test-integration/SKILL.md new file mode 100644 index 000000000..773d0dd26 --- /dev/null +++ b/skills/mem0-test-integration/SKILL.md @@ -0,0 +1,368 @@ +--- +name: mem0-test-integration +description: > + Verify a Mem0 integration produced by /mem0-integrate. Runs in the same + workspace on the same branch (loose coupling) — installs dependencies, + runs the repo's native test suite, then exercises a real end-to-end + smoke flow against the user's API key. Produces a scorecard. + TRIGGER when: user has just run /mem0-integrate and says "verify", + "test the integration", "run /mem0-test-integration", or when a + .mem0-integration/ directory exists and tests have not been run yet + on the current branch. + DO NOT TRIGGER when: the user wants to run general project tests + (defer to the repo's native test command), or when no prior /mem0-integrate + run exists in the current branch (ask them to run /mem0-integrate first). + This skill ONLY catches compile and runtime bugs by design. Logical + integration errors — wrong data stored, wrong time retrieved, wrong + user scoping — are on the human reviewer. +license: Apache-2.0 +metadata: + author: mem0ai + version: "0.1.0" + category: ai-memory + tags: "memory, integration, testing, tdd, platform, oss" + coupling: loose + mem0_tested_versions: "mem0ai (PyPI) >=2.0.0,<3.0.0; mem0ai (npm) >=3.0.0,<4.0.0" +--- + +# mem0-test-integration + +Verifies what `/mem0-integrate` produced. Runs in the same workspace, +on the same feature branch. Loose coupling — fast, catches compile and +runtime bugs, does not catch logical errors. + +## Canonical sources (use these, not ambient knowledge) + +All static checks and smoke-test shapes validate against these URLs. +`WebFetch` each before running step 3. + +- Scope-tagged docs index: https://docs.mem0.ai/llms.txt +- OpenAPI (Platform REST): https://docs.mem0.ai/openapi.json +- Published SDK skill (canonical call patterns): https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0/SKILL.md +- Vercel AI SDK skill (if the target repo uses `@ai-sdk/*`): https://raw.githubusercontent.com/mem0ai/mem0/main/skills/mem0-vercel-ai-sdk/SKILL.md +- SDK source (cross-check version against frontmatter `mem0_tested_versions`): + - Repo root: https://github.com/mem0ai/mem0 + - Python: https://github.com/mem0ai/mem0/tree/main/mem0 + - TypeScript: https://github.com/mem0ai/mem0/tree/main/mem0-ts + +Read the `Delegated skill:` field in `.mem0-integration/plan.md` — if it +names a skill URL, fetch that skill and use its example blocks as the +reference for both static checks (step 3) and the smoke test (step 5). + +## Non-invasiveness contract + +Every check in this skill assumes the integration is **additive and +feature-flagged** (see `/mem0-integrate` "Integration principles"). +Specifically: + +- `product.json` must contain a `feature_flag` field. +- Steps 4–6 run in two passes: + - **Pass A — flag unset.** All pre-existing tests must pass, smoke/E2E + skip. The repo must behave like `main`. Any failure here is a + **hard fail** — do not let the self-heal loop attempt a patch. + - **Pass B — flag set.** New tests must pass, smoke and E2E run. +- If Pass A fails, the scorecard marks `non_invasive: false` and sets + `overall: fail` with a distinct reason code the integrator's heal + loop refuses to touch. + +## Preconditions + +Refuse to start unless ALL of the following are true: + +- `.mem0-integration/` directory exists in the repo root. +- `.mem0-integration/product.json`, `goal.md`, and `plan.md` are readable + and internally consistent (JSON parses, docs non-empty). +- Current branch name begins with `mem0-integrate/` (set by the companion + skill). Prevents accidental runs on unrelated branches. +- Working tree is clean. The skill never modifies source files; any dirty + state means the integration is mid-edit and not ready to verify. +- The same API key the integration used is available in the environment + (`MEM0_API_KEY` for Platform, `OPENAI_API_KEY` for OSS — read which from + `product.json`). Interactive mode asks if missing; CI mode exits 2. + +Exit with a written rationale on any precondition failure. Never attempt +to "fix up" state. + +## Pipeline + +### 1. Read the contract + +Load: + +- `product.json` → which language, which product (Platform vs OSS), which + mem0 version, `write_site`, `read_site`. +- `plan.md` → the mechanical contract (write pattern, read pattern, + preserved behavior). +- `goal.md` → the intent (displayed in the scorecard only; not tested). + +### 2. Install dependencies + +Route by language from `product.json`: + +| Language | Command | +|---|---| +| Python | `pip install -e .` if editable, else `pip install -r requirements.txt`. Then `pip install mem0ai` if not already present at the pinned version. | +| TypeScript / JavaScript | `npm install` (or `pnpm install` / `yarn install` if detected by lockfile). | + +If install fails → exit code 2 with stderr tail. Never move to testing +if dependencies don't resolve. + +### 3. Static sanity checks (fast, local, no API calls) + +- **Import check**: does the write-site file import the expected Mem0 + surface? Authoritative list comes from `## Identify the User's Setup` + in `https://docs.mem0.ai/llms.txt`: + - Platform Python → `from mem0 import MemoryClient` + - Platform TS → `import MemoryClient from "mem0ai"` + - OSS Python → `from mem0 import Memory` + - OSS TS → `import { Memory } from "mem0ai/oss"` + + If `plan.md` names a delegated skill (e.g., Vercel AI), use *that* + skill's import signature instead of the list above. Mismatch → fail + with line number. +- **Version check**: installed `mem0ai` version falls in the range from + this skill's `mem0_tested_versions`. Out of range → warn but continue. +- **Type check** (TS tracks only): run `tsc --noEmit` or `tsup --dts`. + Non-zero → fail. +- **Lint** (if the repo has a linter configured): run the repo's own + lint command. Lint failures from this skill's changes → fail; pre-existing + lint failures → surface as a warning. +- **Eager-init check**: grep the `write_site` and `read_site` files (paths + from `product.json`) for `MemoryClient(` or `Memory(` at module scope — + i.e., not inside a function, method, or class body. `MemoryClient()` + validates the API key in `__init__` (network call) and OSS `Memory()` + can eagerly initialize embedding/LLM providers — module-level + instantiation hits the wire on import and breaks Pass A's test + collection whenever the key is unset. Hit → fail with `file:line` and + the lazy-init guidance from `/mem0-integrate` step 8 constraint #7. + +### 4. Run the repo's native test suite (two passes) + +| Language | Test command (in priority order) | +|---|---| +| Python | `pytest` with the test files from step 5 of the companion skill, else `python -m unittest discover`. | +| TypeScript / JavaScript | `npm test` if defined in package.json; else auto-detect `vitest` or `jest`. | + +**Pass A — `feature_flag` unset.** Run the *entire* pre-existing suite +(excluding the new `test_mem0_*` files). **Must be 100% green.** Any +failure here marks `non_invasive: false` in the scorecard and is +a **hard fail** — the integrator's self-heal loop refuses to touch it. + +**Pass B — `feature_flag` set** (value from `product.json`). Run the +full suite including the new tests. All must pass. + +Isolate integration-introduced failures using `git diff main..HEAD +--name-only`. A test file that exists on `main` and fails only under +the integration branch (flag set *or* unset) counts against the +scorecard regardless of pass. A test file that already failed on `main` +is surfaced as `pre_existing_unrelated` and does not count — but is +still reported so the user can clean it up. + +Capture output to `.mem0-integration/test-stdout-flag-off.log` and +`.mem0-integration/test-stdout-flag-on.log`. Scorecard reports pass/fail +per pass. + +### 5. Smoke test (real API call, shortest round-trip) + +Scripted end-to-end flow tailored to `product.json`. The call shapes +below are the minimal ones; if `plan.md` names a delegated skill, use +*that skill's* minimal example verbatim instead — it is the canonical +shape for the detected stack. + +**Platform (Python):** + + from mem0 import MemoryClient + c = MemoryClient() # uses MEM0_API_KEY + uid = f"mem0-test-integration-{os.urandom(4).hex()}" + c.add([{"role": "user", "content": "I prefer aisle seats"}], user_id=uid) + hits = c.search("seat preference", user_id=uid) + assert any("aisle" in h.get("memory", "") for h in hits), hits + c.delete_all(user_id=uid) # clean up + +**Platform (TS):** same shape with `MemoryClient` from `"mem0ai"`. + +**OSS (Python / TS):** uses `Memory()` / `new Memory()` with default config +(OpenAI LLM via `OPENAI_API_KEY`, local Qdrant). If the repo ships a +`docker-compose.yml` with a Qdrant service, the skill starts it first and +tears it down after. If no backing store is reachable → fail with a +clear message naming the fix. + +The smoke test always uses a **disposable random user_id** prefixed with +`mem0-test-integration-` so a failed cleanup doesn't pollute the user's +real data. A background tidy step deletes any prefix-matching entries +older than 24 hours on the next run. + +Capture output to `.mem0-integration/smoke-stdout.log`. + +### 6. E2E integration test (run the app, exercise the flow) + +Unit tests + smoke prove the SDK works in isolation. This step is the +real signal: **does memory actually appear in the app's user-visible +output when the integration runs end-to-end?** + +Requires `plan.md` to contain an `E2E recipe:` section (authored by +`/mem0-integrate` step 5). If absent → status `skipped` (not `fail`), +note in scorecard that the repo has no runnable entry point. + +Recipe fields the skill reads: + +- `start` — shell command to launch the app using `$PORT` for any network + port. Run in background with stdout/stderr teed to + `.mem0-integration/e2e-app.log`. +- `ready_probe` — how to detect readiness. `url=... status=...` polls an + HTTP endpoint; `log="..."` waits for a substring in `e2e-app.log`; + `sleep=N` waits N seconds (last resort). 60-second hard timeout. +- `compose_services` — optional. If set, bring them up via + `docker compose up -d ` before `start`, tear them down with + `docker compose down` at the end. +- `write_call` — triggers the Mem0 write path exactly once. Output is + captured and surfaced on failure. 60-second hard timeout. +- `write_async_wait_ms` — pause after `write_call` to let async memory + flushes land. Default 0. +- `read_call` — triggers the Mem0 read path. Typically a fresh session + or new request that should surface the stored memory. +- `read_assert` — substring, `regex=...`, or `jsonpath==` + that must appear in `read_call`'s stdout. This is the E2E pass gate. + +Execution order: + +1. Allocate an ephemeral TCP port; export as `PORT`. +2. Set `MEM0_USER_ID` to a disposable `mem0-test-integration-` value + and export it, so the app can use the same scoping the smoke test does + if the recipe wants cleanup. +3. Bring up `compose_services` if named. +4. Run `start` in the background. +5. Poll `ready_probe` until success or 60s timeout. Timeout → fail. +6. Run `write_call`. Non-zero exit → fail (but continue to cleanup). +7. Sleep `write_async_wait_ms`. +8. Run `read_call`. +9. Evaluate `read_assert` against `read_call`'s stdout. Miss → fail. +10. Cleanup (always, even on failure): SIGTERM the app, SIGKILL after + 5s, `docker compose down` if services were started, `delete_all` + memories matching `mem0-test-integration-*` on Platform scenarios. + +On any failure, the scorecard includes: + +- Last 40 lines of `e2e-app.log` +- Full `write_call` output +- Full `read_call` output +- The expected vs actual for `read_assert` + +### 7. Scorecard + +Write `.mem0-integration/scorecard.md` and `.mem0-integration/scorecard.json`: + + { + "timestamp": "2026-04-20T14:03:11Z", + "branch": "mem0-integrate/remember-user-preferences", + "product": "platform", + "language": "python", + "mem0_version": "2.0.0", + "non_invasive": true, + "feature_flag": "MEM0_ENABLED", + "results": { + "install": {"status": "pass", "duration_ms": 12043}, + "static_checks":{"status": "pass", "duration_ms": 812}, + "unit_tests_flag_off": {"status": "pass", "duration_ms": 3920, "count": 47, + "reason": "all pre-existing tests green with flag unset"}, + "unit_tests_flag_on": {"status": "pass", "duration_ms": 4321, "count": 49}, + "smoke_test": {"status": "pass", "duration_ms": 2890, "memory_id": "mem_..."}, + "e2e_test": {"status": "pass", "duration_ms": 14200, + "ready_probe_ms": 3100, "write_exit": 0, + "read_assert_matched": true} + }, + "friction": { + "dependency_install_retries": 0, + "pre_existing_test_failures": 0, + "warnings": ["mem0ai 2.0.0 pinned; consider 2.0.1 for fix X"] + }, + "overall": "pass" + } + +The markdown version is human-readable and includes: + +- Goal doc + plan doc reprinted at top (so reviewers don't have to hunt). +- Each check with pass/fail + log excerpt. +- Friction summary. +- Verbatim warnings from mem0 SDK (if any — e.g., deprecated field usage). +- **Explicit "NOT checked" section** listing what loose coupling misses: + "Whether the stored data is what the user wants stored. Whether search + runs at the right moment. Whether user_id matches the actual session + scope. Human review required." + +### 8. Report + exit + +- Print the scorecard path + overall pass/fail to stdout. +- **Do not commit the scorecard files.** They live in `.mem0-integration/`, + which is gitignored. The user can inspect and optionally pin. +- On fail: print the first failing step's log tail (last 40 lines) and + stop. Do not attempt to fix anything. + +## Artifacts (all under `.mem0-integration/`) + +| File | Purpose | Retention | +|---|---|---| +| `scorecard.md` | Human-readable verdict. | Overwritten per run. | +| `scorecard.json` | Machine-readable verdict. Consumed by the CI scorecard workflow later. | Overwritten per run. | +| `test-stdout-flag-off.log` | Step 4 Pass A (pre-existing suite, flag unset). | Overwritten per run. | +| `test-stdout-flag-on.log` | Step 4 Pass B (full suite, flag set). | Overwritten per run. | +| `smoke-stdout.log` | Full output from step 5. | Overwritten per run. | +| `e2e-app.log` | Background app stdout/stderr from step 6. | Overwritten per run. | +| `e2e-calls.log` | write_call + read_call invocations and outputs. | Overwritten per run. | + +## Modes + +| Mode | Trigger | Behavior | +|---|---|---| +| Interactive (default) | TTY present, `MEM0_TEST_CI` unset | Asks for missing keys, prints friendly summaries. | +| CI | `MEM0_TEST_CI=1` | Keys must be in env, no prompts, non-zero exit on any fail. JSON scorecard goes to stdout's tail for workflow parsing. | + +## Invocation + + /mem0-test-integration # interactive, all steps + /mem0-test-integration --ci # non-interactive + /mem0-test-integration --skip-smoke # no API calls, no E2E + /mem0-test-integration --skip-e2e # unit + smoke only (faster CI) + /mem0-test-integration --only-smoke # just smoke + /mem0-test-integration --only-e2e # just E2E (assumes deps installed) + +Composition: `--skip-*` can stack (`--skip-smoke --skip-e2e` = static + +unit only, zero API cost). `--only-*` is mutually exclusive with all +other flags. + +## Exit codes + +| Code | Meaning | +|---|---| +| 0 | All checks passed. | +| 1 | Precondition failed (no `.mem0-integration/`, wrong branch, dirty tree). | +| 2 | Missing env key (CI mode) or dependency install failure. | +| 3 | Static sanity check failed (wrong import, type error). | +| 4 | Unit tests failed (Pass B — integration itself broken). | +| 5 | Smoke test failed. | +| 6 | E2E test failed (ready_probe timeout, write/read call failed, or read_assert miss). | +| 7 | Non-invasiveness violation: Pass A failed (pre-existing tests broke). Integrator's heal loop refuses to touch this. | +| 8 | Internal error (skill bug — report it). | + +## Explicitly out of scope + +- **Modifying source files.** The skill is read-only against the repo. + If verification exposes a bug, re-run `/mem0-integrate` on the same + goal + plan; do not hand-patch. +- **Fixing broken tests.** Failing unit tests are a signal that the + integration is wrong, not that the tests are wrong. The skill does + not "try a different test." +- **Deep logical correctness.** The E2E step proves "something the user + said earlier comes back later," which is a useful but shallow signal. + It does NOT prove the integration picks the *right* facts to store, + scopes `user_id` correctly across real users, or handles conflict + resolution well. That's human review territory. +- **Self-healing.** This skill never modifies source files. The paired + `/mem0-integrate` skill in its default `--heal` mode consumes the + scorecard produced here and drives its own remediation loop. Exit + code 7 (non-invasiveness violation) is the explicit signal the heal + loop must stop and surface to the user. +- **Cross-branch comparisons.** No `main` baseline diffing. The + scorecard reflects this branch only. +- **Running against production data.** Every smoke test uses a disposable + random user_id and cleans up after. Never touches any other user's data. diff --git a/tests/test_oss_to_platform_migrate.py b/tests/test_oss_to_platform_migrate.py new file mode 100644 index 000000000..8e740bf76 --- /dev/null +++ b/tests/test_oss_to_platform_migrate.py @@ -0,0 +1,838 @@ +from __future__ import annotations + +import json +import os +import subprocess +import threading +from hashlib import sha256 +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import Any +from urllib.parse import parse_qs, urlparse + + +SCRIPT = Path(__file__).resolve().parents[1] / "scripts" / "oss-to-platform-migrate.sh" + + +class MigrationHTTPServer: + def __init__( + self, + *, + ping_emails: dict[str, str] | None = None, + verify_api_key: str = "verified-key", + verify_status: int = 200, + qdrant_api_key: str = "qdrant-key", + qdrant_collection: str = "mem0", + qdrant_pages: list[dict[str, Any]] | None = None, + platform_memories: list[dict[str, Any]] | None = None, + ) -> None: + self.ping_emails = ping_emails or {} + self.verify_api_key = verify_api_key + self.verify_status = verify_status + self.qdrant_api_key = qdrant_api_key + self.qdrant_collection = qdrant_collection + self.qdrant_pages = qdrant_pages or [{"points": [], "next_page_offset": None}] + self.platform_memories = platform_memories or [] + self.requests: list[dict[str, Any]] = [] + self._server = ThreadingHTTPServer(("127.0.0.1", 0), self._handler()) + self.url = f"http://127.0.0.1:{self._server.server_port}" + self._thread = threading.Thread(target=self._server.serve_forever, daemon=True) + + def __enter__(self) -> "MigrationHTTPServer": + self._thread.start() + return self + + def __exit__(self, *_exc: object) -> None: + self._server.shutdown() + self._server.server_close() + self._thread.join(timeout=5) + + def _handler(self) -> type[BaseHTTPRequestHandler]: + owner = self + + class Handler(BaseHTTPRequestHandler): + def log_message(self, _format: str, *_args: object) -> None: + return + + def _read_json(self) -> dict[str, Any]: + length = int(self.headers.get("Content-Length", "0")) + raw = self.rfile.read(length) if length else b"" + if not raw: + return {} + return json.loads(raw.decode("utf-8")) + + def _send_json(self, status: int, payload: dict[str, Any]) -> None: + body = json.dumps(payload).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def _record(self, body: dict[str, Any] | None = None) -> None: + owner.requests.append( + { + "method": self.command, + "path": self.path, + "headers": dict(self.headers), + "body": body or {}, + } + ) + + def do_GET(self) -> None: + self._record() + if self.path == "/v1/ping/": + auth = self.headers.get("Authorization", "") + token = auth.removeprefix("Token ") + email = owner.ping_emails.get(token) + if email: + self._send_json(200, {"user_email": email}) + else: + self._send_json(401, {"detail": "Invalid token"}) + return + if self.path == f"/collections/{owner.qdrant_collection}": + if self.headers.get("api-key") != owner.qdrant_api_key: + self._send_json(401, {"status": {"error": "unauthorized"}}) + else: + self._send_json(200, {"result": {"status": "green"}, "status": "ok"}) + return + self._send_json(404, {"detail": "Not found"}) + + def do_POST(self) -> None: + body = self._read_json() + self._record(body) + parsed = urlparse(self.path) + if self.path == "/posthog": + self._send_json(200, {"ok": True}) + return + if self.path == "/api/v1/auth/email_code/": + self._send_json(200, {"sent": True}) + return + if self.path == "/api/v1/auth/email_code/verify/": + if owner.verify_status != 200: + self._send_json(owner.verify_status, {"error": "bad verification code"}) + else: + self._send_json(200, {"api_key": owner.verify_api_key}) + return + if self.path == f"/collections/{owner.qdrant_collection}/points/scroll": + if self.headers.get("api-key") != owner.qdrant_api_key: + self._send_json(401, {"status": {"error": "unauthorized"}}) + return + offset = body.get("offset") + page_index = int(offset) if offset is not None else 0 + page = owner.qdrant_pages[page_index] + self._send_json( + 200, + { + "result": { + "points": page["points"], + "next_page_offset": page.get("next_page_offset"), + }, + "status": "ok", + }, + ) + return + if parsed.path == "/v3/memories/": + query = parse_qs(parsed.query) + page = int(query.get("page", ["1"])[0]) + page_size = int(query.get("page_size", ["100"])[0]) + filters = body.get("filters") if isinstance(body.get("filters"), dict) else {} + filtered = owner.platform_memories + for key in ("user_id", "agent_id", "run_id"): + if key in filters: + filtered = [memory for memory in filtered if memory.get(key) == filters[key]] + start = (page - 1) * page_size + end = start + page_size + page_results = filtered[start:end] + next_url = ( + f"{owner.url}/v3/memories/?page={page + 1}&page_size={page_size}" + if end < len(filtered) + else None + ) + self._send_json( + 200, + { + "count": len(filtered), + "next": next_url, + "previous": None, + "results": page_results, + }, + ) + return + if parsed.path == "/v3/memories/add/": + memory_id = f"platform-{len(owner.platform_memories) + 1}" + message = (body.get("messages") or [{}])[0] + memory = { + "id": memory_id, + "memory": message.get("content"), + "metadata": body.get("metadata"), + "user_id": body.get("user_id"), + "agent_id": body.get("agent_id"), + "run_id": body.get("run_id"), + } + owner.platform_memories.append(memory) + self._send_json( + 200, + { + "message": "Memories stored successfully", + "status": "SUCCEEDED", + "event_id": "event-1", + "results": [{"id": memory_id, "data": {"memory": message.get("content")}, "event": "ADD"}], + }, + ) + return + self._send_json(404, {"detail": "Not found"}) + + return Handler + + +def alias_marker(anon_id: str, email: str) -> str: + return sha256(f"{anon_id}\0{email}".encode("utf-8")).hexdigest() + + +def write_config(mem0_dir: Path, data: dict[str, Any]) -> None: + mem0_dir.mkdir(parents=True, exist_ok=True) + (mem0_dir / "config.json").write_text(json.dumps(data), encoding="utf-8") + + +def read_config(mem0_dir: Path) -> dict[str, Any]: + return json.loads((mem0_dir / "config.json").read_text(encoding="utf-8")) + + +def run_migration_script( + tmp_path: Path, + server: MigrationHTTPServer, + *args: str, + config: dict[str, Any] | None = None, + raw_config: str | None = None, +) -> tuple[subprocess.CompletedProcess[str], Path]: + mem0_dir = tmp_path / "mem0" + if config is not None: + write_config(mem0_dir, config) + if raw_config is not None: + mem0_dir.mkdir(parents=True, exist_ok=True) + (mem0_dir / "config.json").write_text(raw_config, encoding="utf-8") + + env = os.environ.copy() + env.update( + { + "MEM0_DIR": str(mem0_dir), + "MEM0_MIGRATE_TELEMETRY_URL": f"{server.url}/posthog", + } + ) + env.pop("MEM0_API_KEY", None) + env.pop("MEM0_BASE_URL", None) + + result = subprocess.run( + ["bash", str(SCRIPT), "--auth-only", "--base-url", server.url, *args], + capture_output=True, + text=True, + env=env, + start_new_session=True, + timeout=20, + check=False, + ) + return result, mem0_dir + + +def run_export_script( + tmp_path: Path, + server: MigrationHTTPServer, + *args: str, + config: dict[str, Any] | None = None, + qdrant_api_key: str = "qdrant-key", +) -> tuple[subprocess.CompletedProcess[str], Path, Path]: + mem0_dir = tmp_path / "mem0" + output_path = tmp_path / "export.json" + if config is not None: + write_config(mem0_dir, config) + + env = os.environ.copy() + env.update( + { + "MEM0_DIR": str(mem0_dir), + "MEM0_MIGRATE_TELEMETRY_URL": f"{server.url}/posthog", + "QDRANT_API_KEY": qdrant_api_key, + } + ) + env.pop("MEM0_API_KEY", None) + env.pop("MEM0_BASE_URL", None) + + result = subprocess.run( + [ + "bash", + str(SCRIPT), + "--export-only", + "--qdrant-url", + server.url, + "--qdrant-collection", + server.qdrant_collection, + "--output", + str(output_path), + *args, + ], + capture_output=True, + text=True, + env=env, + start_new_session=True, + timeout=20, + check=False, + ) + return result, mem0_dir, output_path + + +def run_import_script( + tmp_path: Path, + server: MigrationHTTPServer, + input_path: Path, + *args: str, + config: dict[str, Any] | None = None, + api_key: str = "import-key", +) -> tuple[subprocess.CompletedProcess[str], Path]: + mem0_dir = tmp_path / "mem0" + if config is not None: + write_config(mem0_dir, config) + + env = os.environ.copy() + env.update( + { + "MEM0_DIR": str(mem0_dir), + "MEM0_MIGRATE_TELEMETRY_URL": f"{server.url}/posthog", + "MEM0_API_KEY": api_key, + } + ) + env.pop("MEM0_BASE_URL", None) + + result = subprocess.run( + [ + "bash", + str(SCRIPT), + "--import-only", + "--base-url", + server.url, + "--input", + str(input_path), + *args, + ], + capture_output=True, + text=True, + env=env, + start_new_session=True, + timeout=20, + check=False, + ) + return result, mem0_dir + + +def run_full_script( + tmp_path: Path, + server: MigrationHTTPServer, + *args: str, + config: dict[str, Any] | None = None, + qdrant_api_key: str = "qdrant-key", +) -> tuple[subprocess.CompletedProcess[str], Path, Path]: + mem0_dir = tmp_path / "mem0" + output_path = tmp_path / "full-export.json" + if config is not None: + write_config(mem0_dir, config) + + env = os.environ.copy() + env.update( + { + "MEM0_DIR": str(mem0_dir), + "MEM0_MIGRATE_TELEMETRY_URL": f"{server.url}/posthog", + "QDRANT_API_KEY": qdrant_api_key, + } + ) + env.pop("MEM0_API_KEY", None) + env.pop("MEM0_BASE_URL", None) + + result = subprocess.run( + [ + "bash", + str(SCRIPT), + "--base-url", + server.url, + "--qdrant-url", + server.url, + "--qdrant-collection", + server.qdrant_collection, + "--output", + str(output_path), + *args, + ], + capture_output=True, + text=True, + env=env, + start_new_session=True, + timeout=20, + check=False, + ) + return result, mem0_dir, output_path + + +def posthog_events(server: MigrationHTTPServer) -> list[dict[str, Any]]: + return [request["body"] for request in server.requests if request["path"] == "/posthog"] + + +def test_existing_api_key_authenticates_and_stitches_ids(tmp_path: Path) -> None: + config = { + "user_id": "oss-123", + "platform": {"api_key": "stored-key", "base_url": "https://api.mem0.ai"}, + "telemetry": {"anonymous_id": "cli-456"}, + } + + with MigrationHTTPServer(ping_emails={"stored-key": "bob@example.com"}) as server: + result, mem0_dir = run_migration_script(tmp_path, server, "--yes", config=config) + + assert result.returncode == 0, result.stderr + assert "Authenticated as bob@example.com" in result.stdout + assert not any(request["path"] == "/api/v1/auth/email_code/verify/" for request in server.requests) + + updated = read_config(mem0_dir) + assert updated["platform"] == config["platform"] + assert alias_marker("oss-123", "bob@example.com") in updated["telemetry"]["aliased_pairs"] + assert alias_marker("cli-456", "bob@example.com") in updated["telemetry"]["aliased_pairs"] + + events = posthog_events(server) + event_names = [event["event"] for event in events] + assert "oss.migrate.started" in event_names + assert "oss.migrate.authenticated" in event_names + assert event_names.count("$identify") == 2 + + authenticated = next(event for event in events if event["event"] == "oss.migrate.authenticated") + assert authenticated["distinct_id"] == "bob@example.com" + assert authenticated["properties"]["local_anonymous_id"] == "oss-123" + assert authenticated["properties"]["authenticated_email"] == "bob@example.com" + + +def test_email_code_authenticates_without_persisting_credentials(tmp_path: Path) -> None: + with MigrationHTTPServer(ping_emails={"verified-key": "alice@example.com"}) as server: + result, mem0_dir = run_migration_script( + tmp_path, + server, + "--email", + "Alice@Example.COM", + "--code", + "123456", + ) + + assert result.returncode == 0, result.stderr + assert "Authenticated as alice@example.com" in result.stdout + + verify_request = next(request for request in server.requests if request["path"] == "/api/v1/auth/email_code/verify/") + assert verify_request["body"] == {"email": "alice@example.com", "code": "123456"} + assert not any(request["path"] == "/api/v1/auth/email_code/" for request in server.requests) + + updated = read_config(mem0_dir) + assert "api_key" not in updated.get("platform", {}) + assert "user_email" not in updated.get("platform", {}) + assert updated["user_id"] + assert alias_marker(updated["user_id"], "alice@example.com") in updated["telemetry"]["aliased_pairs"] + + events = posthog_events(server) + assert [event["event"] for event in events].count("$identify") == 1 + authenticated = next(event for event in events if event["event"] == "oss.migrate.authenticated") + assert authenticated["properties"]["auth_method"] == "email_code" + + +def test_invalid_stored_key_falls_back_to_email_code(tmp_path: Path) -> None: + config = { + "user_id": "oss-fallback", + "platform": {"api_key": "bad-key", "base_url": "https://api.mem0.ai"}, + } + + with MigrationHTTPServer(ping_emails={"verified-key": "new@example.com"}) as server: + result, mem0_dir = run_migration_script( + tmp_path, + server, + "--email", + "new@example.com", + "--code", + "123456", + config=config, + ) + + assert result.returncode == 0, result.stderr + assert "Stored Mem0 Platform API key is invalid or expired" in result.stdout + assert "Authenticated as new@example.com" in result.stdout + + ping_tokens = [ + request["headers"]["Authorization"].removeprefix("Token ") + for request in server.requests + if request["path"] == "/v1/ping/" + ] + assert ping_tokens == ["bad-key", "verified-key"] + + updated = read_config(mem0_dir) + assert updated["platform"] == config["platform"] + assert alias_marker("oss-fallback", "new@example.com") in updated["telemetry"]["aliased_pairs"] + + +def test_email_code_failure_reports_failed_telemetry(tmp_path: Path) -> None: + with MigrationHTTPServer(verify_status=400) as server: + result, mem0_dir = run_migration_script( + tmp_path, + server, + "--email", + "fail@example.com", + "--code", + "bad", + ) + + assert result.returncode == 1 + assert "Verification failed: bad verification code" in result.stderr + + updated = read_config(mem0_dir) + assert "telemetry" not in updated or "aliased_pairs" not in updated["telemetry"] + + events = posthog_events(server) + event_names = [event["event"] for event in events] + assert "oss.migrate.started" in event_names + assert "oss.migrate.failed" in event_names + assert "$identify" not in event_names + failed = next(event for event in events if event["event"] == "oss.migrate.failed") + assert "Verification failed" in failed["properties"]["error"] + + +def test_malformed_config_does_not_crash_and_authenticates(tmp_path: Path) -> None: + with MigrationHTTPServer(ping_emails={"verified-key": "malformed@example.com"}) as server: + result, mem0_dir = run_migration_script( + tmp_path, + server, + "--email", + "malformed@example.com", + "--code", + "123456", + raw_config="{not valid json", + ) + + assert result.returncode == 0, result.stderr + assert "Authenticated as malformed@example.com" in result.stdout + + updated = read_config(mem0_dir) + assert updated["user_id"] + assert alias_marker(updated["user_id"], "malformed@example.com") in updated["telemetry"]["aliased_pairs"] + + +def test_weird_telemetry_shape_does_not_crash(tmp_path: Path) -> None: + config = {"user_id": "oss-weird-telemetry", "telemetry": "not-an-object"} + + with MigrationHTTPServer(ping_emails={"verified-key": "weird@example.com"}) as server: + result, mem0_dir = run_migration_script( + tmp_path, + server, + "--email", + "weird@example.com", + "--code", + "123456", + config=config, + ) + + assert result.returncode == 0, result.stderr + updated = read_config(mem0_dir) + assert isinstance(updated["telemetry"], dict) + assert alias_marker("oss-weird-telemetry", "weird@example.com") in updated["telemetry"]["aliased_pairs"] + + +def test_missing_python3_prints_clear_shell_error(tmp_path: Path) -> None: + env = os.environ.copy() + env["PATH"] = str(tmp_path) + + result = subprocess.run( + ["/bin/bash", str(SCRIPT), "--help"], + capture_output=True, + text=True, + env=env, + timeout=20, + check=False, + ) + + assert result.returncode == 1 + assert "python3 is required to run the Mem0 migration" in result.stderr + + +def test_curl_piped_help_works() -> None: + result = subprocess.run( + ["bash", "-c", f"curl -fsSL file://{SCRIPT} | bash -s -- --help"], + capture_output=True, + text=True, + timeout=20, + check=False, + ) + + assert result.returncode == 0, result.stderr + assert "Migrate Python OSS hosted-Qdrant memories" in result.stdout + + +def test_export_qdrant_memories_to_json_without_vectors_or_api_key(tmp_path: Path) -> None: + pages = [ + { + "points": [ + { + "id": "point-1", + "vector": [0.1, 0.2], + "payload": { + "data": "User likes dark mode", + "hash": "hash-1", + "created_at": "2026-05-01T00:00:00Z", + "updated_at": "2026-05-01T00:00:00Z", + "user_id": "alice", + "agent_id": "agent-1", + "run_id": "run-1", + "actor_id": "actor-1", + "role": "user", + "topic": "preferences", + "text_lemmatized": "user like dark mode", + }, + } + ], + "next_page_offset": 1, + }, + { + "points": [ + { + "id": "point-2", + "vector": [0.3, 0.4], + "payload": { + "data": "User prefers concise answers", + "hash": "hash-2", + "user_id": "alice", + "metadata_note": "extra", + }, + } + ], + "next_page_offset": None, + }, + ] + + with MigrationHTTPServer(qdrant_pages=pages) as server: + result, _mem0_dir, output_path = run_export_script( + tmp_path, + server, + "--user-id", + "alice", + "--qdrant-page-size", + "1", + config={"user_id": "oss-export-user"}, + ) + + assert result.returncode == 0, result.stderr + assert "Exported 2 memories" in result.stdout + + artifact = json.loads(output_path.read_text(encoding="utf-8")) + assert artifact["kind"] == "mem0_oss_qdrant_export" + assert artifact["source"]["sdk"] == "python" + assert artifact["source"]["vector_store"] == "qdrant" + assert artifact["source"]["storage"] == "hosted" + assert artifact["source"]["filters"]["user_id"] == "alice" + assert artifact["record_count"] == 2 + assert artifact["local_anonymous_id"] == "oss-export-user" + + first = artifact["records"][0] + assert first["id"] == "point-1" + assert first["memory"] == "User likes dark mode" + assert first["hash"] == "hash-1" + assert first["user_id"] == "alice" + assert first["agent_id"] == "agent-1" + assert first["run_id"] == "run-1" + assert first["actor_id"] == "actor-1" + assert first["role"] == "user" + assert first["metadata"] == {"topic": "preferences"} + assert all("vector" not in record for record in artifact["records"]) + assert "qdrant-key" not in output_path.read_text(encoding="utf-8") + + scroll_requests = [request for request in server.requests if request["path"].endswith("/points/scroll")] + assert len(scroll_requests) == 2 + assert scroll_requests[0]["body"]["with_vector"] is False + assert scroll_requests[0]["body"]["filter"] == {"must": [{"key": "user_id", "match": {"value": "alice"}}]} + assert scroll_requests[1]["body"]["offset"] == 1 + + +def test_export_requires_scope_or_all(tmp_path: Path) -> None: + with MigrationHTTPServer() as server: + result, _mem0_dir, output_path = run_export_script(tmp_path, server) + + assert result.returncode == 1 + assert "Export requires --user-id, --agent-id, --run-id, or --all" in result.stderr + assert not output_path.exists() + assert not any(request["path"].endswith("/points/scroll") for request in server.requests) + + +def test_export_all_uses_no_qdrant_filter(tmp_path: Path) -> None: + with MigrationHTTPServer(qdrant_pages=[{"points": [], "next_page_offset": None}]) as server: + result, _mem0_dir, output_path = run_export_script(tmp_path, server, "--all") + + assert result.returncode == 0, result.stderr + artifact = json.loads(output_path.read_text(encoding="utf-8")) + assert artifact["record_count"] == 0 + assert artifact["records"] == [] + + scroll_request = next(request for request in server.requests if request["path"].endswith("/points/scroll")) + assert "filter" not in scroll_request["body"] + + +def test_export_invalid_qdrant_credentials_fail_clearly(tmp_path: Path) -> None: + with MigrationHTTPServer(qdrant_api_key="correct-key") as server: + result, _mem0_dir, output_path = run_export_script( + tmp_path, + server, + "--user-id", + "alice", + qdrant_api_key="wrong-key", + ) + + assert result.returncode == 1 + assert "Qdrant authentication failed" in result.stderr + assert not output_path.exists() + + +def test_import_platform_memories_from_export_json(tmp_path: Path) -> None: + input_path = tmp_path / "import.json" + input_path.write_text( + json.dumps( + { + "source": {"sdk": "python", "vector_store": "qdrant", "collection": "mem0_test"}, + "records": [ + { + "id": "local-1", + "memory": "User likes barbecue", + "hash": "hash-1", + "created_at": "2026-05-07T00:00:00Z", + "user_id": "alice", + "metadata": {"topic": "food"}, + } + ], + } + ), + encoding="utf-8", + ) + + with MigrationHTTPServer(ping_emails={"import-key": "alice@example.com"}) as server: + result, _mem0_dir = run_import_script(tmp_path, server, input_path) + + assert result.returncode == 0, result.stderr + assert "Imported: 1" in result.stdout + assert "Failed: 0" in result.stdout + + add_request = next(request for request in server.requests if request["path"] == "/v3/memories/add/") + body = add_request["body"] + assert body["messages"] == [{"role": "user", "content": "User likes barbecue"}] + assert body["user_id"] == "alice" + assert body["infer"] is False + assert body["source"] == "migration" + assert body["timestamp"] == 1778112000 + assert body["metadata"]["topic"] == "food" + assert body["metadata"]["mem0_migration_source"] == "python_oss_qdrant" + assert body["metadata"]["mem0_migration_collection"] == "mem0_test" + assert body["metadata"]["mem0_migration_local_id"] == "local-1" + assert body["metadata"]["mem0_migration_local_hash"] == "hash-1" + + events = posthog_events(server) + assert "oss.migrate.completed" in [event["event"] for event in events] + + +def test_import_skips_existing_identical_memory(tmp_path: Path) -> None: + input_path = tmp_path / "import.json" + source = {"sdk": "python", "vector_store": "qdrant", "collection": "mem0_test"} + record = {"id": "local-1", "memory": "User likes barbecue", "hash": "hash-1", "user_id": "alice"} + input_path.write_text(json.dumps({"source": source, "records": [record]}), encoding="utf-8") + import_key = sha256("python:qdrant:mem0_test:local-1".encode("utf-8")).hexdigest() + + existing = [ + { + "id": "platform-1", + "memory": "User likes barbecue", + "user_id": "alice", + "metadata": { + "mem0_migration_import_key": import_key, + "mem0_migration_local_hash": "hash-1", + }, + } + ] + + with MigrationHTTPServer(ping_emails={"import-key": "alice@example.com"}, platform_memories=existing) as server: + result, _mem0_dir = run_import_script(tmp_path, server, input_path) + + assert result.returncode == 0, result.stderr + assert "Imported: 0" in result.stdout + assert "Skipped existing identical: 1" in result.stdout + assert "Changed existing: 0" in result.stdout + assert not any(request["path"] == "/v3/memories/add/" for request in server.requests) + + +def test_import_reports_changed_existing_without_update_or_add(tmp_path: Path) -> None: + input_path = tmp_path / "import.json" + source = {"sdk": "python", "vector_store": "qdrant", "collection": "mem0_test"} + record = {"id": "local-1", "memory": "User likes brisket", "hash": "hash-new", "user_id": "alice"} + input_path.write_text(json.dumps({"source": source, "records": [record]}), encoding="utf-8") + import_key = sha256("python:qdrant:mem0_test:local-1".encode("utf-8")).hexdigest() + + existing = [ + { + "id": "platform-1", + "memory": "User likes barbecue", + "user_id": "alice", + "metadata": { + "mem0_migration_import_key": import_key, + "mem0_migration_local_hash": "hash-old", + }, + } + ] + + with MigrationHTTPServer(ping_emails={"import-key": "alice@example.com"}, platform_memories=existing) as server: + result, _mem0_dir = run_import_script(tmp_path, server, input_path) + + assert result.returncode == 0, result.stderr + assert "Imported: 0" in result.stdout + assert "Skipped existing identical: 0" in result.stdout + assert "Changed existing: 1" in result.stdout + assert not any(request["path"] == "/v3/memories/add/" for request in server.requests) + review_path_line = next(line for line in result.stdout.splitlines() if line.startswith("Review file: ")) + review_path = Path(review_path_line.removeprefix("Review file: ")) + review = json.loads(review_path.read_text(encoding="utf-8")) + assert review["records"][0]["status"] == "changed_existing" + assert review["records"][0]["platform_memory_id"] == "platform-1" + + +def test_full_flow_auth_export_and_imports_memories(tmp_path: Path) -> None: + qdrant_pages = [ + { + "points": [ + { + "id": "point-1", + "payload": { + "data": "User likes barbecue", + "hash": "hash-1", + "created_at": "2026-05-07T00:00:00Z", + "user_id": "alice", + }, + } + ], + "next_page_offset": None, + } + ] + + with MigrationHTTPServer(ping_emails={"verified-key": "alice@example.com"}, qdrant_pages=qdrant_pages) as server: + result, _mem0_dir, output_path = run_full_script( + tmp_path, + server, + "--email", + "alice@example.com", + "--code", + "123456", + "--user-id", + "alice", + ) + + assert result.returncode == 0, result.stderr + assert "Phase 1/3: Authenticate with Mem0 Platform" in result.stdout + assert "Phase 2/3: Export Python OSS memories from hosted Qdrant" in result.stdout + assert "Phase 3/3: Import memories into Mem0 Platform" in result.stdout + assert "Imported: 1" in result.stdout + assert output_path.exists() + assert any(request["path"] == "/v3/memories/add/" for request in server.requests) + events = posthog_events(server) + event_names = [event["event"] for event in events] + assert "oss.migrate.authenticated" in event_names + assert "oss.migrate.completed" in event_names diff --git a/tests/test_project.py b/tests/test_project.py new file mode 100644 index 000000000..9f2d9b7a2 --- /dev/null +++ b/tests/test_project.py @@ -0,0 +1,97 @@ +"""Tests for ``mem0.client.project.Project.update`` — focused on the +parameter-passthrough surface. + +Verifies the kwarg → JSON payload mapping for every supported field +(``custom_instructions``, ``custom_categories``, ``retrieval_criteria``, +``multilingual``, ``decay``), the ValueError when no field is +provided, and the URL/method shape. The HTTP layer is mocked. +""" + +from unittest.mock import MagicMock, patch + +import pytest + + +@pytest.fixture +def project(): + """Build a ``Project`` with a mocked httpx client. + + Bypasses ``MemoryClient`` so the test stays focused on + ``Project.update`` payload construction. + """ + http = MagicMock() + http.patch.return_value = MagicMock( + json=lambda: {"message": "Updated"}, + raise_for_status=lambda: None, + ) + with patch("mem0.client.project.capture_client_event"): + from mem0.client.project import Project + + proj = Project(client=http, org_id="org1", project_id="proj1") + yield proj, http + + +def _patch_payload(http): + """Return the JSON body sent on the last PATCH, stripped of the SDK's + standard auth params (``org_id``, ``project_id``) that ``_prepare_params`` + injects on every request.""" + assert http.patch.called, "expected a PATCH call" + _, kwargs = http.patch.call_args + body = dict(kwargs.get("json", {})) + body.pop("org_id", None) + body.pop("project_id", None) + return body + + +class TestProjectUpdateDecay: + def test_decay_true_sent_in_payload(self, project): + proj, http = project + proj.update(decay=True) + assert _patch_payload(http) == {"decay": True} + + def test_decay_false_sent_in_payload(self, project): + """Explicit ``False`` must round-trip — not be filtered as falsy.""" + proj, http = project + proj.update(decay=False) + assert _patch_payload(http) == {"decay": False} + + def test_decay_combined_with_multilingual(self, project): + proj, http = project + proj.update(multilingual=True, decay=True) + assert _patch_payload(http) == { + "multilingual": True, + "decay": True, + } + + def test_decay_omitted_when_none(self, project): + """When the caller doesn't pass ``decay``, it must not appear in + the payload — backwards compatible with pre-decay callers.""" + proj, http = project + proj.update(multilingual=False) + payload = _patch_payload(http) + assert payload == {"multilingual": False} + assert "decay" not in payload + + def test_no_args_raises_with_decay_in_message(self, project): + proj, _ = project + with pytest.raises(ValueError, match=r"decay"): + proj.update() + + def test_url_targets_project_endpoint(self, project): + proj, http = project + proj.update(decay=True) + args, _ = http.patch.call_args + assert args[0] == "/api/v1/orgs/organizations/org1/projects/proj1/" + + +class TestProjectUpdateBackwardsCompat: + def test_multilingual_only_still_works(self, project): + """Pre-decay callers (multilingual only) keep working unchanged.""" + proj, http = project + proj.update(multilingual=True) + assert _patch_payload(http) == {"multilingual": True} + + def test_custom_instructions_only_still_works(self, project): + proj, http = project + proj.update(custom_instructions="be concise") + assert _patch_payload(http) == {"custom_instructions": "be concise"} diff --git a/tests/test_telemetry_aliasing.py b/tests/test_telemetry_aliasing.py new file mode 100644 index 000000000..bd6fcf0ee --- /dev/null +++ b/tests/test_telemetry_aliasing.py @@ -0,0 +1,411 @@ +"""Tests for PostHog identity stitching: anon → email alias on MemoryClient init. + +Covers the four matrix cases (OSS-only, CLI-only, both, already-aliased) plus +failure modes: missing config, malformed JSON, read-only filesystem, +broken posthog client. Telemetry must never raise. +""" + +import importlib +import json +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + + +@pytest.fixture +def tmp_mem0_dir(tmp_path, monkeypatch): + """Point the mem0 setup module at a tempdir for the duration of the test.""" + monkeypatch.setenv("MEM0_DIR", str(tmp_path)) + # Reload setup so module-level mem0_dir picks up the env var. + import mem0.memory.setup as setup_module + + importlib.reload(setup_module) + yield tmp_path + # Restore default state. + monkeypatch.delenv("MEM0_DIR", raising=False) + importlib.reload(setup_module) + + +def _write_config(tmp_path: Path, payload: dict) -> Path: + config_path = tmp_path / "config.json" + config_path.write_text(json.dumps(payload)) + return config_path + + +# ─── setup_config idempotency ──────────────────────────────────────────────── + + +class TestSetupConfigIdempotent: + def test_creates_config_when_missing(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + setup_module.setup_config() + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert "user_id" in config and config["user_id"] + + def test_backfills_user_id_when_only_telemetry_present(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config( + tmp_mem0_dir, + {"telemetry": {"anonymous_id": "cli-anon-abc123"}}, + ) + setup_module.setup_config() + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert config.get("user_id"), "user_id must be backfilled for CLI-first users" + assert config["telemetry"]["anonymous_id"] == "cli-anon-abc123" + + def test_does_not_overwrite_existing_user_id(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config(tmp_mem0_dir, {"user_id": "existing-uuid"}) + setup_module.setup_config() + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert config["user_id"] == "existing-uuid" + + def test_handles_malformed_json(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + (tmp_mem0_dir / "config.json").write_text("{not json") + setup_module.setup_config() # must not raise + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert "user_id" in config + + +# ─── read_anon_ids ─────────────────────────────────────────────────────────── + + +class TestReadAnonIds: + def test_returns_oss_only(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config(tmp_mem0_dir, {"user_id": "oss-uuid"}) + anon = setup_module.read_anon_ids() + assert anon == {"oss": "oss-uuid", "cli": None, "aliased_pairs": []} + + def test_returns_cli_only(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config( + tmp_mem0_dir, + {"telemetry": {"anonymous_id": "cli-anon-123"}}, + ) + anon = setup_module.read_anon_ids() + assert anon == {"oss": None, "cli": "cli-anon-123", "aliased_pairs": []} + + def test_returns_both(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config( + tmp_mem0_dir, + { + "user_id": "oss-uuid", + "telemetry": {"anonymous_id": "cli-anon-123", "aliased_pairs": ["pair-marker"]}, + }, + ) + anon = setup_module.read_anon_ids() + assert anon == { + "oss": "oss-uuid", + "cli": "cli-anon-123", + "aliased_pairs": ["pair-marker"], + } + + def test_no_config_returns_all_none(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + anon = setup_module.read_anon_ids() + assert anon == {"oss": None, "cli": None, "aliased_pairs": []} + + def test_malformed_json_does_not_raise(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + (tmp_mem0_dir / "config.json").write_text("{not json") + anon = setup_module.read_anon_ids() + assert anon == {"oss": None, "cli": None, "aliased_pairs": []} + + +# ─── mark_aliased ──────────────────────────────────────────────────────────── + + +class TestMarkAliased: + def test_writes_aliased_pair_preserving_other_fields(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config( + tmp_mem0_dir, + { + "user_id": "oss-uuid", + "telemetry": {"anonymous_id": "cli-anon-123"}, + }, + ) + setup_module.mark_aliased("oss-uuid", "user@example.com") + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert config["user_id"] == "oss-uuid" + assert config["telemetry"]["anonymous_id"] == "cli-anon-123" + assert len(config["telemetry"]["aliased_pairs"]) == 1 + assert setup_module.is_aliased("oss-uuid", "user@example.com") + + def test_creates_telemetry_section_when_missing(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config(tmp_mem0_dir, {"user_id": "oss-uuid"}) + setup_module.mark_aliased("oss-uuid", "user@example.com") + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert len(config["telemetry"]["aliased_pairs"]) == 1 + + def test_tracks_each_pair_independently(self, tmp_mem0_dir): + import mem0.memory.setup as setup_module + + _write_config(tmp_mem0_dir, {"user_id": "oss-uuid"}) + setup_module.mark_aliased("oss-uuid", "user@example.com") + assert setup_module.is_aliased("oss-uuid", "user@example.com") + assert not setup_module.is_aliased("new-uuid", "user@example.com") + assert not setup_module.is_aliased("oss-uuid", "other@example.com") + + +# ─── capture_identify ──────────────────────────────────────────────────────── + + +class TestCaptureIdentify: + def test_fires_identify_with_anon_distinct_id(self): + import mem0.memory.telemetry as telemetry_module + + with patch.object(telemetry_module, "MEM0_TELEMETRY", True): + with patch("mem0.memory.telemetry.Posthog") as mock_posthog_cls: + at = telemetry_module.AnonymousTelemetry() + at.capture_identify("anon-123", "user@example.com") + mock_ph = mock_posthog_cls.return_value + mock_ph.capture.assert_called_once() + _, kwargs = mock_ph.capture.call_args + assert kwargs["distinct_id"] == "user@example.com" + assert kwargs["event"] == "$identify" + assert kwargs["properties"]["$anon_distinct_id"] == "anon-123" + + def test_skips_when_anon_equals_email(self): + import mem0.memory.telemetry as telemetry_module + + with patch.object(telemetry_module, "MEM0_TELEMETRY", True): + with patch("mem0.memory.telemetry.Posthog") as mock_posthog_cls: + at = telemetry_module.AnonymousTelemetry() + at.capture_identify("user@example.com", "user@example.com") + mock_posthog_cls.return_value.capture.assert_not_called() + + def test_skips_when_inputs_empty(self): + import mem0.memory.telemetry as telemetry_module + + with patch.object(telemetry_module, "MEM0_TELEMETRY", True): + with patch("mem0.memory.telemetry.Posthog") as mock_posthog_cls: + at = telemetry_module.AnonymousTelemetry() + at.capture_identify("", "user@example.com") + at.capture_identify("anon-123", "") + mock_posthog_cls.return_value.capture.assert_not_called() + + def test_noop_when_telemetry_disabled(self): + import mem0.memory.telemetry as telemetry_module + + with patch.object(telemetry_module, "MEM0_TELEMETRY", False): + at = telemetry_module.AnonymousTelemetry() + at.capture_identify("anon-123", "user@example.com") # must not raise + assert at.posthog is None + + def test_does_not_raise_on_posthog_error(self): + import mem0.memory.telemetry as telemetry_module + + with patch.object(telemetry_module, "MEM0_TELEMETRY", True): + with patch("mem0.memory.telemetry.Posthog") as mock_posthog_cls: + mock_posthog_cls.return_value.capture.side_effect = RuntimeError("boom") + at = telemetry_module.AnonymousTelemetry() + at.capture_identify("anon-123", "user@example.com") # must not raise + + def test_identify_is_in_lifecycle_events(self): + """$identify must bypass the 90% sampling drop.""" + import mem0.memory.telemetry as telemetry_module + + assert "$identify" in telemetry_module._LIFECYCLE_EVENTS + + +# ─── _maybe_alias_anon_to_email integration ────────────────────────────────── + + +class TestMaybeAliasAnonToEmail: + """Test the alias helper in isolation by mocking out the config readers + and the telemetry client, since module-level setup_config() side effects + make end-to-end fixturing awkward.""" + + def test_fires_identify_for_oss_uuid(self): + from mem0.client import main as client_main + + with ( + patch.object( + client_main, + "read_anon_ids", + return_value={"oss": "oss-uuid", "cli": None, "aliased_pairs": []}, + ), + patch.object(client_main, "is_aliased", return_value=False), + patch.object(client_main, "mark_aliased") as mark, + patch.object(client_main, "client_telemetry") as telemetry, + ): + telemetry.capture_identify.return_value = True + client_main._maybe_alias_anon_to_email("user@example.com") + telemetry.capture_identify.assert_called_once_with("oss-uuid", "user@example.com") + mark.assert_called_once_with("oss-uuid", "user@example.com") + + def test_fires_identify_for_cli_anon(self): + from mem0.client import main as client_main + + with ( + patch.object( + client_main, + "read_anon_ids", + return_value={"oss": None, "cli": "cli-anon-xyz", "aliased_pairs": []}, + ), + patch.object(client_main, "is_aliased", return_value=False), + patch.object(client_main, "mark_aliased"), + patch.object(client_main, "client_telemetry") as telemetry, + ): + telemetry.capture_identify.return_value = True + client_main._maybe_alias_anon_to_email("user@example.com") + telemetry.capture_identify.assert_called_once_with("cli-anon-xyz", "user@example.com") + + def test_fires_identify_for_both_anon_ids(self): + from mem0.client import main as client_main + + with ( + patch.object( + client_main, + "read_anon_ids", + return_value={"oss": "oss-uuid", "cli": "cli-anon", "aliased_pairs": []}, + ), + patch.object(client_main, "is_aliased", return_value=False), + patch.object(client_main, "mark_aliased"), + patch.object(client_main, "client_telemetry") as telemetry, + ): + telemetry.capture_identify.return_value = True + client_main._maybe_alias_anon_to_email("user@example.com") + assert telemetry.capture_identify.call_count == 2 + calls = {c.args for c in telemetry.capture_identify.call_args_list} + assert ("oss-uuid", "user@example.com") in calls + assert ("cli-anon", "user@example.com") in calls + + def test_skips_when_pair_already_aliased(self): + from mem0.client import main as client_main + + with ( + patch.object( + client_main, + "read_anon_ids", + return_value={"oss": "oss-uuid", "cli": None, "aliased_pairs": ["pair-marker"]}, + ), + patch.object(client_main, "is_aliased", return_value=True), + patch.object(client_main, "mark_aliased") as mark, + patch.object(client_main, "client_telemetry") as telemetry, + ): + client_main._maybe_alias_anon_to_email("user@example.com") + telemetry.capture_identify.assert_not_called() + mark.assert_not_called() + + def test_skips_when_email_invalid(self): + from mem0.client import main as client_main + + with patch.object(client_main, "client_telemetry") as telemetry: + client_main._maybe_alias_anon_to_email(None) + client_main._maybe_alias_anon_to_email("") + client_main._maybe_alias_anon_to_email("not-an-email") + telemetry.capture_identify.assert_not_called() + + def test_skips_when_telemetry_disabled(self): + """When client_telemetry.posthog is None (MEM0_TELEMETRY=false), do nothing — + no fs read, no fs write, no event. Re-enabling telemetry later must still alias.""" + from mem0.client import main as client_main + + disabled = MagicMock() + disabled.posthog = None + with ( + patch.object(client_main, "client_telemetry", disabled), + patch.object(client_main, "read_anon_ids") as read, + patch.object(client_main, "mark_aliased") as mark, + ): + client_main._maybe_alias_anon_to_email("user@example.com") + read.assert_not_called() + mark.assert_not_called() + disabled.capture_identify.assert_not_called() + + def test_does_not_raise_on_telemetry_failure(self): + from mem0.client import main as client_main + + mock_telemetry = MagicMock() + mock_telemetry.capture_identify.side_effect = RuntimeError("boom") + with ( + patch.object( + client_main, + "read_anon_ids", + return_value={"oss": "oss-uuid", "cli": None, "aliased_pairs": []}, + ), + patch.object(client_main, "is_aliased", return_value=False), + patch.object(client_main, "mark_aliased") as mark, + patch.object(client_main, "client_telemetry", mock_telemetry), + ): + client_main._maybe_alias_anon_to_email("user@example.com") # must not raise + mark.assert_not_called() + + def test_skips_anon_id_equal_to_email(self): + """Defensive: if the anon_id somehow already is the email, don't self-alias.""" + from mem0.client import main as client_main + + with ( + patch.object( + client_main, + "read_anon_ids", + return_value={"oss": "user@example.com", "cli": None, "aliased_pairs": []}, + ), + patch.object(client_main, "is_aliased", return_value=False), + patch.object(client_main, "mark_aliased"), + patch.object(client_main, "client_telemetry") as telemetry, + ): + client_main._maybe_alias_anon_to_email("user@example.com") + telemetry.capture_identify.assert_not_called() + + def test_does_not_raise_on_read_failure(self): + """If read_anon_ids itself raises (e.g. IO error), helper must swallow it.""" + from mem0.client import main as client_main + + with ( + patch.object(client_main, "read_anon_ids", side_effect=OSError("fs broken")), + patch.object(client_main, "client_telemetry") as telemetry, + ): + client_main._maybe_alias_anon_to_email("user@example.com") # must not raise + telemetry.capture_identify.assert_not_called() + + +# ─── End-to-end idempotency through real config ────────────────────────────── + + +class TestEndToEndIdempotency: + """Verify the real config flow: two consecutive _maybe_alias_anon_to_email + calls fire $identify exactly once thanks to the persisted pair marker.""" + + def test_second_call_is_noop_after_pair_marker_persisted(self, tmp_mem0_dir): + # Pre-populate config with an OSS user_id only. + _write_config(tmp_mem0_dir, {"user_id": "oss-uuid"}) + # Reload setup so it uses the tempdir, then reload client.main so it + # picks up the freshly-loaded read_anon_ids/mark_aliased bindings. + import mem0.memory.setup as setup_module + + importlib.reload(setup_module) + from mem0.client import main as client_main + + importlib.reload(client_main) + + with patch.object(client_main, "client_telemetry") as telemetry: + telemetry.capture_identify.return_value = True + client_main._maybe_alias_anon_to_email("user@example.com") + first_call_count = telemetry.capture_identify.call_count + assert first_call_count >= 1 + + # Second call should hit the aliased_pairs short-circuit. + client_main._maybe_alias_anon_to_email("user@example.com") + assert telemetry.capture_identify.call_count == first_call_count + + config = json.loads((tmp_mem0_dir / "config.json").read_text()) + assert len(config["telemetry"]["aliased_pairs"]) == 1